Show how a word is pronounced, Classical Persian first

Wiktionary carries pronunciation for 45,000 of these headwords and the build
was throwing it away. It is now kept: 101,306 entries of IPA, each tagged with
the variety it belongs to, which matters here because a word in a 14th-century
ghazal was not said the way Tehran says it now — عشق is /ˈʔiʃq/ in Classical
Persian and [ʔeʃɢ̥] in Iran today, and the app leads with the former.

The IPA's dots and stress marks are the syllable breakdown. Urdu Wiktionary
adds its own, in Urdu script: the fully vowelled spelling عِشْق and the split
عِش + قوں, both pulled out of its wikitext.

Costs 7.7 MB of database, taking it to 94 MB.

Verified on an API 36 emulator: tapping عشق shows the Classical Persian IPA,
the Urdu syllable split, the vowelled spelling and two modern variants above
the definitions.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Anas Rashid 2026-10-04 17:48:12 +02:00
parent b16d020ce0
commit 0df736fd89
5 changed files with 92 additions and 4 deletions

Binary file not shown.

View File

@ -11,13 +11,19 @@ import java.text.Normalizer
/** One definition, and where it came from, so the credit stays attached to the text. */ /** One definition, and where it came from, so the credit stays attached to the text. */
data class Definition(val word: String, val gloss: String, val source: String) data class Definition(val word: String, val gloss: String, val source: String)
/**
* How a word sounds. [label] is the variety it belongs to — Classical Persian, Dari, Standard
* Urdu — because a word in a 14th-century ghazal was not said the way Tehran says it now.
*/
data class Pronunciation(val text: String, val label: String)
/** /**
* Word lookup over a bundled SQLite built from Wiktionary (CC BY-SA 3.0) and Daneshjoo. * Word lookup over a bundled SQLite built from Wiktionary (CC BY-SA 3.0) and Daneshjoo.
* See `tools/build_dictionary.py`; the asset ships gzipped and is unpacked once on first use. * See `tools/build_dictionary.py`; the asset ships gzipped and is unpacked once on first use.
*/ */
object Dictionary { object Dictionary {
/** Guards against a copy interrupted half-way leaving an unopenable file behind. */ /** Guards against a copy interrupted half-way leaving an unopenable file behind. */
private const val ASSET_BYTES = 86663168L private const val ASSET_BYTES = 94384128L
// Not a .gz: the build packager silently gunzips those and drops the extension, which left // Not a .gz: the build packager silently gunzips those and drops the extension, which left
// the asset under a different name than the code was opening. // the asset under a different name than the code was opening.
@ -46,6 +52,37 @@ object Dictionary {
} }
} }
/**
* How [raw] is pronounced, Classical Persian first: this is an app for poetry written long
* before modern Tehrani vowels, and the IPA's dots and stress marks are the syllable
* breakdown that goes with it.
*/
suspend fun pronunciations(raw: String, limit: Int = 5): List<Pronunciation> =
withContext(Dispatchers.IO) {
val database = open() ?: return@withContext emptyList()
val word = normalise(raw).takeIf { it.isNotEmpty() } ?: return@withContext emptyList()
database.rawQuery(
"""
SELECT text, label FROM pron WHERE word = ?
ORDER BY CASE
WHEN label LIKE 'Classical%' THEN 0
WHEN source = 'urwiktionary' THEN 1
WHEN source = 'wiktionary-fa' THEN 2
WHEN source = 'wiktionary-ur' THEN 3
ELSE 4
END
LIMIT ?
""",
arrayOf(word, limit.toString()),
).use { cursor ->
buildList {
while (cursor.moveToNext()) {
add(Pronunciation(cursor.getString(0), cursor.getString(1)))
}
}
}
}
/** /**
* Looks a word up, widening the search until something matches: * Looks a word up, widening the search until something matches:
* *

View File

@ -30,6 +30,7 @@ import androidx.compose.ui.unit.dp
import com.ganjoor.android.R import com.ganjoor.android.R
import com.ganjoor.android.data.Definition import com.ganjoor.android.data.Definition
import com.ganjoor.android.data.Dictionary import com.ganjoor.android.data.Dictionary
import com.ganjoor.android.data.Pronunciation
import com.ganjoor.android.ui.theme.readingStyle import com.ganjoor.android.ui.theme.readingStyle
/** English prose inside an otherwise right-to-left sheet. */ /** English prose inside an otherwise right-to-left sheet. */
@ -69,6 +70,9 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
val definitions by produceState<List<Definition>?>(null, current) { val definitions by produceState<List<Definition>?>(null, current) {
value = Dictionary.lookup(current) value = Dictionary.lookup(current)
} }
val sounds by produceState(emptyList<Pronunciation>(), current) {
value = Dictionary.pronunciations(current)
}
val suggestions by produceState(emptyList<String>(), current, definitions) { val suggestions by produceState(emptyList<String>(), current, definitions) {
value = if (definitions?.isEmpty() == true) Dictionary.suggest(current) else emptyList() value = if (definitions?.isEmpty() == true) Dictionary.suggest(current) else emptyList()
} }
@ -88,6 +92,25 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
style = readingStyle(prefs.font, prefs.fontSize, prefs.fontWeight.weight), style = readingStyle(prefs.font, prefs.fontSize, prefs.fontWeight.weight),
modifier = Modifier.fillMaxWidth(), modifier = Modifier.fillMaxWidth(),
) )
if (sounds.isNotEmpty()) {
FlowRow(horizontalArrangement = Arrangement.spacedBy(12.dp)) {
sounds.forEach { sound ->
Column {
// IPA is Latin-script, so it reads left to right whatever the UI does
LeftToRight {
Text(sound.text, style = MaterialTheme.typography.bodyMedium)
}
if (sound.label.isNotBlank()) {
Text(
text = sound.label,
style = MaterialTheme.typography.labelSmall,
color = MaterialTheme.colorScheme.onSurfaceVariant,
)
}
}
}
}
}
HorizontalDivider() HorizontalDivider()
when { when {

View File

@ -21,7 +21,7 @@ Nothing in Ganjoor's repositories restricts AI-assisted use.
## Dictionary ## Dictionary
Four sources, each row in the database tagged with the one it came from so the app can name it Five sources, each row in the database tagged with the one it came from so the app can name it
and so any of them can be dropped without rebuilding the others. and so any of them can be dropped without rebuilding the others.
| Source | Direction | Licence | | Source | Direction | Licence |
@ -29,9 +29,13 @@ and so any of them can be dropped without rebuilding the others.
| [Wiktionary](https://en.wiktionary.org) | Persian → English | CC BY-SA 3.0 | | [Wiktionary](https://en.wiktionary.org) | Persian → English | CC BY-SA 3.0 |
| [Wiktionary](https://en.wiktionary.org) | Urdu → English | CC BY-SA 3.0 | | [Wiktionary](https://en.wiktionary.org) | Urdu → English | CC BY-SA 3.0 |
| [Urdu Wiktionary](https://ur.wiktionary.org) | Urdu → Urdu | CC BY-SA 3.0 | | [Urdu Wiktionary](https://ur.wiktionary.org) | Urdu → Urdu | CC BY-SA 3.0 |
| [Wiktionary](https://en.wiktionary.org) | Arabic → English | CC BY-SA 3.0 |
| [Daneshjoo](https://github.com/0xdolan/Daneshjoo) | Persian → English | Repository states MIT | | [Daneshjoo](https://github.com/0xdolan/Daneshjoo) | Persian → English | Repository states MIT |
Because three of the four are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**. Pronunciation — IPA with the variety it belongs to, and Urdu Wiktionary's vowelled spelling and
syllable split — comes from the same Wiktionary exports and carries the same licence.
Because four of the five are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**.
Attribution is shown in the app beside every definition. See [`../tools/README.md`](../tools/README.md) Attribution is shown in the app beside every definition. See [`../tools/README.md`](../tools/README.md)
to rebuild it. to rebuild it.

View File

@ -29,9 +29,20 @@ c = sqlite3.connect(db)
c.executescript(""" c.executescript("""
CREATE TABLE entry (word TEXT NOT NULL, display TEXT NOT NULL, gloss TEXT NOT NULL, source TEXT NOT NULL); CREATE TABLE entry (word TEXT NOT NULL, display TEXT NOT NULL, gloss TEXT NOT NULL, source TEXT NOT NULL);
CREATE TABLE form (form TEXT NOT NULL, lemma TEXT NOT NULL); CREATE TABLE form (form TEXT NOT NULL, lemma TEXT NOT NULL);
CREATE TABLE pron (word TEXT NOT NULL, text TEXT NOT NULL, label TEXT NOT NULL, source TEXT NOT NULL);
""") """)
entries, forms = [], set() entries, forms, prons = [], set(), set()
def collect_sounds(word, entry, source):
"""IPA with whatever dialect it belongs to. Classical Persian matters most here: this is an
app for poetry written a long time before modern Tehrani vowels."""
for sound in entry.get('sounds') or []:
ipa = (sound.get('ipa') or '').strip()
if not ipa:
continue
tags = [t for t in (sound.get('tags') or []) if t not in ('formal',)]
prons.add((normalise(word), ipa, ' '.join(tags), source))
for line in open('fa.jsonl', encoding='utf-8'): for line in open('fa.jsonl', encoding='utf-8'):
try: e = json.loads(line) try: e = json.loads(line)
@ -43,6 +54,7 @@ for line in open('fa.jsonl', encoding='utf-8'):
pos = e.get('pos') or '' pos = e.get('pos') or ''
gloss = '; '.join(dict.fromkeys(gs))[:600] gloss = '; '.join(dict.fromkeys(gs))[:600]
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa')) entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa'))
collect_sounds(word, e, 'wiktionary-fa')
for f in e.get('forms', []): for f in e.get('forms', []):
t = f.get('form') t = f.get('form')
if t and t != word and not t.startswith('-') and len(t) > 1: if t and t != word and not t.startswith('-') and len(t) > 1:
@ -63,6 +75,7 @@ for line in open('ur.jsonl', encoding='utf-8'):
pos = e.get('pos') or '' pos = e.get('pos') or ''
gloss = '; '.join(dict.fromkeys(gs))[:600] gloss = '; '.join(dict.fromkeys(gs))[:600]
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur')) entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur'))
collect_sounds(word, e, 'wiktionary-ur')
for f in e.get('forms', []): for f in e.get('forms', []):
t = f.get('form') t = f.get('form')
if t and t != word and not t.startswith('-') and len(t) > 1: if t and t != word and not t.startswith('-') and len(t) > 1:
@ -82,6 +95,7 @@ if os.path.exists('ar.jsonl'):
pos = e.get('pos') or '' pos = e.get('pos') or ''
gloss = '; '.join(dict.fromkeys(gs))[:600] gloss = '; '.join(dict.fromkeys(gs))[:600]
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar')) entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar'))
collect_sounds(word, e, 'wiktionary-ar')
for f in e.get('forms', []): for f in e.get('forms', []):
t = f.get('form') t = f.get('form')
if t and t != word and not t.startswith('-') and len(t) > 1: if t and t != word and not t.startswith('-') and len(t) > 1:
@ -136,13 +150,23 @@ if os.path.exists('urwikt.xml'):
gloss = ' '.join(kept).strip()[:300] gloss = ' '.join(kept).strip()[:300]
if len(gloss) > 3: if len(gloss) > 3:
entries.append((normalise(title), title, gloss, 'urwiktionary')) entries.append((normalise(title), title, gloss, 'urwiktionary'))
# {عِشْق} is the fully vowelled spelling; "عِش + قوں" is the syllable split
vowelled = re.search(r'\{([\u0600-\u06FF\u064B-\u0652 ]{2,40})\}', t)
if vowelled:
prons.add((normalise(title), vowelled.group(1).strip(), 'اردو', 'urwiktionary'))
syllables = re.search(r'\{([\u0600-\u06FF\u064B-\u0652]+(?: \+ [\u0600-\u06FF\u064B-\u0652]+)+[^}]*)\}', t)
if syllables:
prons.add((normalise(title), syllables.group(1).strip(), 'ہجے', 'urwiktionary'))
print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)") print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)")
print(f"pronunciations: {len(prons)}")
c.executemany("INSERT INTO pron VALUES (?,?,?,?)", sorted(prons))
c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries) c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries)
c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms)) c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms))
c.executescript(""" c.executescript("""
CREATE INDEX idx_entry_word ON entry(word); CREATE INDEX idx_entry_word ON entry(word);
CREATE INDEX idx_form_form ON form(form); CREATE INDEX idx_form_form ON form(form);
CREATE INDEX idx_pron_word ON pron(word);
""") """)
c.commit() c.commit()
c.execute("VACUUM") c.execute("VACUUM")