Show how a word is pronounced, Classical Persian first
Wiktionary carries pronunciation for 45,000 of these headwords and the build was throwing it away. It is now kept: 101,306 entries of IPA, each tagged with the variety it belongs to, which matters here because a word in a 14th-century ghazal was not said the way Tehran says it now — عشق is /ˈʔiʃq/ in Classical Persian and [ʔeʃɢ̥] in Iran today, and the app leads with the former. The IPA's dots and stress marks are the syllable breakdown. Urdu Wiktionary adds its own, in Urdu script: the fully vowelled spelling عِشْق and the split عِش + قوں, both pulled out of its wikitext. Costs 7.7 MB of database, taking it to 94 MB. Verified on an API 36 emulator: tapping عشق shows the Classical Persian IPA, the Urdu syllable split, the vowelled spelling and two modern variants above the definitions. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
b16d020ce0
commit
0df736fd89
Binary file not shown.
@ -11,13 +11,19 @@ import java.text.Normalizer
|
||||
/** One definition, and where it came from, so the credit stays attached to the text. */
|
||||
data class Definition(val word: String, val gloss: String, val source: String)
|
||||
|
||||
/**
|
||||
* How a word sounds. [label] is the variety it belongs to — Classical Persian, Dari, Standard
|
||||
* Urdu — because a word in a 14th-century ghazal was not said the way Tehran says it now.
|
||||
*/
|
||||
data class Pronunciation(val text: String, val label: String)
|
||||
|
||||
/**
|
||||
* Word lookup over a bundled SQLite built from Wiktionary (CC BY-SA 3.0) and Daneshjoo.
|
||||
* See `tools/build_dictionary.py`; the asset ships gzipped and is unpacked once on first use.
|
||||
*/
|
||||
object Dictionary {
|
||||
/** Guards against a copy interrupted half-way leaving an unopenable file behind. */
|
||||
private const val ASSET_BYTES = 86663168L
|
||||
private const val ASSET_BYTES = 94384128L
|
||||
|
||||
// Not a .gz: the build packager silently gunzips those and drops the extension, which left
|
||||
// the asset under a different name than the code was opening.
|
||||
@ -46,6 +52,37 @@ object Dictionary {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* How [raw] is pronounced, Classical Persian first: this is an app for poetry written long
|
||||
* before modern Tehrani vowels, and the IPA's dots and stress marks are the syllable
|
||||
* breakdown that goes with it.
|
||||
*/
|
||||
suspend fun pronunciations(raw: String, limit: Int = 5): List<Pronunciation> =
|
||||
withContext(Dispatchers.IO) {
|
||||
val database = open() ?: return@withContext emptyList()
|
||||
val word = normalise(raw).takeIf { it.isNotEmpty() } ?: return@withContext emptyList()
|
||||
database.rawQuery(
|
||||
"""
|
||||
SELECT text, label FROM pron WHERE word = ?
|
||||
ORDER BY CASE
|
||||
WHEN label LIKE 'Classical%' THEN 0
|
||||
WHEN source = 'urwiktionary' THEN 1
|
||||
WHEN source = 'wiktionary-fa' THEN 2
|
||||
WHEN source = 'wiktionary-ur' THEN 3
|
||||
ELSE 4
|
||||
END
|
||||
LIMIT ?
|
||||
""",
|
||||
arrayOf(word, limit.toString()),
|
||||
).use { cursor ->
|
||||
buildList {
|
||||
while (cursor.moveToNext()) {
|
||||
add(Pronunciation(cursor.getString(0), cursor.getString(1)))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Looks a word up, widening the search until something matches:
|
||||
*
|
||||
|
||||
@ -30,6 +30,7 @@ import androidx.compose.ui.unit.dp
|
||||
import com.ganjoor.android.R
|
||||
import com.ganjoor.android.data.Definition
|
||||
import com.ganjoor.android.data.Dictionary
|
||||
import com.ganjoor.android.data.Pronunciation
|
||||
import com.ganjoor.android.ui.theme.readingStyle
|
||||
|
||||
/** English prose inside an otherwise right-to-left sheet. */
|
||||
@ -69,6 +70,9 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
|
||||
val definitions by produceState<List<Definition>?>(null, current) {
|
||||
value = Dictionary.lookup(current)
|
||||
}
|
||||
val sounds by produceState(emptyList<Pronunciation>(), current) {
|
||||
value = Dictionary.pronunciations(current)
|
||||
}
|
||||
val suggestions by produceState(emptyList<String>(), current, definitions) {
|
||||
value = if (definitions?.isEmpty() == true) Dictionary.suggest(current) else emptyList()
|
||||
}
|
||||
@ -88,6 +92,25 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
|
||||
style = readingStyle(prefs.font, prefs.fontSize, prefs.fontWeight.weight),
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
)
|
||||
if (sounds.isNotEmpty()) {
|
||||
FlowRow(horizontalArrangement = Arrangement.spacedBy(12.dp)) {
|
||||
sounds.forEach { sound ->
|
||||
Column {
|
||||
// IPA is Latin-script, so it reads left to right whatever the UI does
|
||||
LeftToRight {
|
||||
Text(sound.text, style = MaterialTheme.typography.bodyMedium)
|
||||
}
|
||||
if (sound.label.isNotBlank()) {
|
||||
Text(
|
||||
text = sound.label,
|
||||
style = MaterialTheme.typography.labelSmall,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
HorizontalDivider()
|
||||
|
||||
when {
|
||||
|
||||
@ -21,7 +21,7 @@ Nothing in Ganjoor's repositories restricts AI-assisted use.
|
||||
|
||||
## Dictionary
|
||||
|
||||
Four sources, each row in the database tagged with the one it came from so the app can name it
|
||||
Five sources, each row in the database tagged with the one it came from so the app can name it
|
||||
and so any of them can be dropped without rebuilding the others.
|
||||
|
||||
| Source | Direction | Licence |
|
||||
@ -29,9 +29,13 @@ and so any of them can be dropped without rebuilding the others.
|
||||
| [Wiktionary](https://en.wiktionary.org) | Persian → English | CC BY-SA 3.0 |
|
||||
| [Wiktionary](https://en.wiktionary.org) | Urdu → English | CC BY-SA 3.0 |
|
||||
| [Urdu Wiktionary](https://ur.wiktionary.org) | Urdu → Urdu | CC BY-SA 3.0 |
|
||||
| [Wiktionary](https://en.wiktionary.org) | Arabic → English | CC BY-SA 3.0 |
|
||||
| [Daneshjoo](https://github.com/0xdolan/Daneshjoo) | Persian → English | Repository states MIT |
|
||||
|
||||
Because three of the four are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**.
|
||||
Pronunciation — IPA with the variety it belongs to, and Urdu Wiktionary's vowelled spelling and
|
||||
syllable split — comes from the same Wiktionary exports and carries the same licence.
|
||||
|
||||
Because four of the five are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**.
|
||||
Attribution is shown in the app beside every definition. See [`../tools/README.md`](../tools/README.md)
|
||||
to rebuild it.
|
||||
|
||||
|
||||
@ -29,9 +29,20 @@ c = sqlite3.connect(db)
|
||||
c.executescript("""
|
||||
CREATE TABLE entry (word TEXT NOT NULL, display TEXT NOT NULL, gloss TEXT NOT NULL, source TEXT NOT NULL);
|
||||
CREATE TABLE form (form TEXT NOT NULL, lemma TEXT NOT NULL);
|
||||
CREATE TABLE pron (word TEXT NOT NULL, text TEXT NOT NULL, label TEXT NOT NULL, source TEXT NOT NULL);
|
||||
""")
|
||||
|
||||
entries, forms = [], set()
|
||||
entries, forms, prons = [], set(), set()
|
||||
|
||||
def collect_sounds(word, entry, source):
|
||||
"""IPA with whatever dialect it belongs to. Classical Persian matters most here: this is an
|
||||
app for poetry written a long time before modern Tehrani vowels."""
|
||||
for sound in entry.get('sounds') or []:
|
||||
ipa = (sound.get('ipa') or '').strip()
|
||||
if not ipa:
|
||||
continue
|
||||
tags = [t for t in (sound.get('tags') or []) if t not in ('formal',)]
|
||||
prons.add((normalise(word), ipa, ' '.join(tags), source))
|
||||
|
||||
for line in open('fa.jsonl', encoding='utf-8'):
|
||||
try: e = json.loads(line)
|
||||
@ -43,6 +54,7 @@ for line in open('fa.jsonl', encoding='utf-8'):
|
||||
pos = e.get('pos') or ''
|
||||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa'))
|
||||
collect_sounds(word, e, 'wiktionary-fa')
|
||||
for f in e.get('forms', []):
|
||||
t = f.get('form')
|
||||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||||
@ -63,6 +75,7 @@ for line in open('ur.jsonl', encoding='utf-8'):
|
||||
pos = e.get('pos') or ''
|
||||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur'))
|
||||
collect_sounds(word, e, 'wiktionary-ur')
|
||||
for f in e.get('forms', []):
|
||||
t = f.get('form')
|
||||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||||
@ -82,6 +95,7 @@ if os.path.exists('ar.jsonl'):
|
||||
pos = e.get('pos') or ''
|
||||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar'))
|
||||
collect_sounds(word, e, 'wiktionary-ar')
|
||||
for f in e.get('forms', []):
|
||||
t = f.get('form')
|
||||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||||
@ -136,13 +150,23 @@ if os.path.exists('urwikt.xml'):
|
||||
gloss = ' '.join(kept).strip()[:300]
|
||||
if len(gloss) > 3:
|
||||
entries.append((normalise(title), title, gloss, 'urwiktionary'))
|
||||
# {عِشْق} is the fully vowelled spelling; "عِش + قوں" is the syllable split
|
||||
vowelled = re.search(r'\{([\u0600-\u06FF\u064B-\u0652 ]{2,40})\}', t)
|
||||
if vowelled:
|
||||
prons.add((normalise(title), vowelled.group(1).strip(), 'اردو', 'urwiktionary'))
|
||||
syllables = re.search(r'\{([\u0600-\u06FF\u064B-\u0652]+(?: \+ [\u0600-\u06FF\u064B-\u0652]+)+[^}]*)\}', t)
|
||||
if syllables:
|
||||
prons.add((normalise(title), syllables.group(1).strip(), 'ہجے', 'urwiktionary'))
|
||||
print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)")
|
||||
|
||||
print(f"pronunciations: {len(prons)}")
|
||||
c.executemany("INSERT INTO pron VALUES (?,?,?,?)", sorted(prons))
|
||||
c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries)
|
||||
c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms))
|
||||
c.executescript("""
|
||||
CREATE INDEX idx_entry_word ON entry(word);
|
||||
CREATE INDEX idx_form_form ON form(form);
|
||||
CREATE INDEX idx_pron_word ON pron(word);
|
||||
""")
|
||||
c.commit()
|
||||
c.execute("VACUUM")
|
||||
|
||||
Loading…
Reference in New Issue
Block a user