diff --git a/app/src/main/assets/dictionary.db b/app/src/main/assets/dictionary.db index c6c6d55..92e8958 100644 Binary files a/app/src/main/assets/dictionary.db and b/app/src/main/assets/dictionary.db differ diff --git a/app/src/main/java/com/ganjoor/android/data/Dictionary.kt b/app/src/main/java/com/ganjoor/android/data/Dictionary.kt index a53ff9e..8ce40c9 100644 --- a/app/src/main/java/com/ganjoor/android/data/Dictionary.kt +++ b/app/src/main/java/com/ganjoor/android/data/Dictionary.kt @@ -11,13 +11,19 @@ import java.text.Normalizer /** One definition, and where it came from, so the credit stays attached to the text. */ data class Definition(val word: String, val gloss: String, val source: String) +/** + * How a word sounds. [label] is the variety it belongs to — Classical Persian, Dari, Standard + * Urdu — because a word in a 14th-century ghazal was not said the way Tehran says it now. + */ +data class Pronunciation(val text: String, val label: String) + /** * Word lookup over a bundled SQLite built from Wiktionary (CC BY-SA 3.0) and Daneshjoo. * See `tools/build_dictionary.py`; the asset ships gzipped and is unpacked once on first use. */ object Dictionary { /** Guards against a copy interrupted half-way leaving an unopenable file behind. */ - private const val ASSET_BYTES = 86663168L + private const val ASSET_BYTES = 94384128L // Not a .gz: the build packager silently gunzips those and drops the extension, which left // the asset under a different name than the code was opening. @@ -46,6 +52,37 @@ object Dictionary { } } + /** + * How [raw] is pronounced, Classical Persian first: this is an app for poetry written long + * before modern Tehrani vowels, and the IPA's dots and stress marks are the syllable + * breakdown that goes with it. + */ + suspend fun pronunciations(raw: String, limit: Int = 5): List = + withContext(Dispatchers.IO) { + val database = open() ?: return@withContext emptyList() + val word = normalise(raw).takeIf { it.isNotEmpty() } ?: return@withContext emptyList() + database.rawQuery( + """ + SELECT text, label FROM pron WHERE word = ? + ORDER BY CASE + WHEN label LIKE 'Classical%' THEN 0 + WHEN source = 'urwiktionary' THEN 1 + WHEN source = 'wiktionary-fa' THEN 2 + WHEN source = 'wiktionary-ur' THEN 3 + ELSE 4 + END + LIMIT ? + """, + arrayOf(word, limit.toString()), + ).use { cursor -> + buildList { + while (cursor.moveToNext()) { + add(Pronunciation(cursor.getString(0), cursor.getString(1))) + } + } + } + } + /** * Looks a word up, widening the search until something matches: * diff --git a/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt b/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt index 3815952..35360c2 100644 --- a/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt +++ b/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt @@ -30,6 +30,7 @@ import androidx.compose.ui.unit.dp import com.ganjoor.android.R import com.ganjoor.android.data.Definition import com.ganjoor.android.data.Dictionary +import com.ganjoor.android.data.Pronunciation import com.ganjoor.android.ui.theme.readingStyle /** English prose inside an otherwise right-to-left sheet. */ @@ -69,6 +70,9 @@ fun WordSheet(word: String, onDismiss: () -> Unit) { val definitions by produceState?>(null, current) { value = Dictionary.lookup(current) } + val sounds by produceState(emptyList(), current) { + value = Dictionary.pronunciations(current) + } val suggestions by produceState(emptyList(), current, definitions) { value = if (definitions?.isEmpty() == true) Dictionary.suggest(current) else emptyList() } @@ -88,6 +92,25 @@ fun WordSheet(word: String, onDismiss: () -> Unit) { style = readingStyle(prefs.font, prefs.fontSize, prefs.fontWeight.weight), modifier = Modifier.fillMaxWidth(), ) + if (sounds.isNotEmpty()) { + FlowRow(horizontalArrangement = Arrangement.spacedBy(12.dp)) { + sounds.forEach { sound -> + Column { + // IPA is Latin-script, so it reads left to right whatever the UI does + LeftToRight { + Text(sound.text, style = MaterialTheme.typography.bodyMedium) + } + if (sound.label.isNotBlank()) { + Text( + text = sound.label, + style = MaterialTheme.typography.labelSmall, + color = MaterialTheme.colorScheme.onSurfaceVariant, + ) + } + } + } + } + } HorizontalDivider() when { diff --git a/licenses/README.md b/licenses/README.md index 002a7ad..ab5758a 100644 --- a/licenses/README.md +++ b/licenses/README.md @@ -21,7 +21,7 @@ Nothing in Ganjoor's repositories restricts AI-assisted use. ## Dictionary -Four sources, each row in the database tagged with the one it came from so the app can name it +Five sources, each row in the database tagged with the one it came from so the app can name it and so any of them can be dropped without rebuilding the others. | Source | Direction | Licence | @@ -29,9 +29,13 @@ and so any of them can be dropped without rebuilding the others. | [Wiktionary](https://en.wiktionary.org) | Persian → English | CC BY-SA 3.0 | | [Wiktionary](https://en.wiktionary.org) | Urdu → English | CC BY-SA 3.0 | | [Urdu Wiktionary](https://ur.wiktionary.org) | Urdu → Urdu | CC BY-SA 3.0 | +| [Wiktionary](https://en.wiktionary.org) | Arabic → English | CC BY-SA 3.0 | | [Daneshjoo](https://github.com/0xdolan/Daneshjoo) | Persian → English | Repository states MIT | -Because three of the four are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**. +Pronunciation — IPA with the variety it belongs to, and Urdu Wiktionary's vowelled spelling and +syllable split — comes from the same Wiktionary exports and carries the same licence. + +Because four of the five are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**. Attribution is shown in the app beside every definition. See [`../tools/README.md`](../tools/README.md) to rebuild it. diff --git a/tools/build_dictionary.py b/tools/build_dictionary.py index 57f7c7e..6271063 100644 --- a/tools/build_dictionary.py +++ b/tools/build_dictionary.py @@ -29,9 +29,20 @@ c = sqlite3.connect(db) c.executescript(""" CREATE TABLE entry (word TEXT NOT NULL, display TEXT NOT NULL, gloss TEXT NOT NULL, source TEXT NOT NULL); CREATE TABLE form (form TEXT NOT NULL, lemma TEXT NOT NULL); +CREATE TABLE pron (word TEXT NOT NULL, text TEXT NOT NULL, label TEXT NOT NULL, source TEXT NOT NULL); """) -entries, forms = [], set() +entries, forms, prons = [], set(), set() + +def collect_sounds(word, entry, source): + """IPA with whatever dialect it belongs to. Classical Persian matters most here: this is an + app for poetry written a long time before modern Tehrani vowels.""" + for sound in entry.get('sounds') or []: + ipa = (sound.get('ipa') or '').strip() + if not ipa: + continue + tags = [t for t in (sound.get('tags') or []) if t not in ('formal',)] + prons.add((normalise(word), ipa, ' '.join(tags), source)) for line in open('fa.jsonl', encoding='utf-8'): try: e = json.loads(line) @@ -43,6 +54,7 @@ for line in open('fa.jsonl', encoding='utf-8'): pos = e.get('pos') or '' gloss = '; '.join(dict.fromkeys(gs))[:600] entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa')) + collect_sounds(word, e, 'wiktionary-fa') for f in e.get('forms', []): t = f.get('form') if t and t != word and not t.startswith('-') and len(t) > 1: @@ -63,6 +75,7 @@ for line in open('ur.jsonl', encoding='utf-8'): pos = e.get('pos') or '' gloss = '; '.join(dict.fromkeys(gs))[:600] entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur')) + collect_sounds(word, e, 'wiktionary-ur') for f in e.get('forms', []): t = f.get('form') if t and t != word and not t.startswith('-') and len(t) > 1: @@ -82,6 +95,7 @@ if os.path.exists('ar.jsonl'): pos = e.get('pos') or '' gloss = '; '.join(dict.fromkeys(gs))[:600] entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar')) + collect_sounds(word, e, 'wiktionary-ar') for f in e.get('forms', []): t = f.get('form') if t and t != word and not t.startswith('-') and len(t) > 1: @@ -136,13 +150,23 @@ if os.path.exists('urwikt.xml'): gloss = ' '.join(kept).strip()[:300] if len(gloss) > 3: entries.append((normalise(title), title, gloss, 'urwiktionary')) + # {عِشْق} is the fully vowelled spelling; "عِش + قوں" is the syllable split + vowelled = re.search(r'\{([\u0600-\u06FF\u064B-\u0652 ]{2,40})\}', t) + if vowelled: + prons.add((normalise(title), vowelled.group(1).strip(), 'اردو', 'urwiktionary')) + syllables = re.search(r'\{([\u0600-\u06FF\u064B-\u0652]+(?: \+ [\u0600-\u06FF\u064B-\u0652]+)+[^}]*)\}', t) + if syllables: + prons.add((normalise(title), syllables.group(1).strip(), 'ہجے', 'urwiktionary')) print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)") +print(f"pronunciations: {len(prons)}") +c.executemany("INSERT INTO pron VALUES (?,?,?,?)", sorted(prons)) c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries) c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms)) c.executescript(""" CREATE INDEX idx_entry_word ON entry(word); CREATE INDEX idx_form_form ON form(form); +CREATE INDEX idx_pron_word ON pron(word); """) c.commit() c.execute("VACUUM")