diff --git a/app/src/main/assets/dictionary.db b/app/src/main/assets/dictionary.db index c6ddab9..88feb42 100644 Binary files a/app/src/main/assets/dictionary.db and b/app/src/main/assets/dictionary.db differ diff --git a/app/src/main/java/com/ganjoor/android/data/Dictionary.kt b/app/src/main/java/com/ganjoor/android/data/Dictionary.kt index 002ff0b..ec3e3ec 100644 --- a/app/src/main/java/com/ganjoor/android/data/Dictionary.kt +++ b/app/src/main/java/com/ganjoor/android/data/Dictionary.kt @@ -17,7 +17,7 @@ data class Definition(val word: String, val gloss: String, val source: String) */ object Dictionary { /** Guards against a copy interrupted half-way leaving an unopenable file behind. */ - private const val ASSET_BYTES = 22_298_624L + private const val ASSET_BYTES = 27742208L // Not a .gz: the build packager silently gunzips those and drops the extension, which left // the asset under a different name than the code was opening. @@ -95,6 +95,47 @@ object Dictionary { } } + /** + * Headwords that look like [raw], for when nothing matched exactly — a misread letter, an + * unusual spelling, or a word the dictionary simply spells differently. + * + * Candidates come from an index range scan on the first letters, then are ranked by how + * many letters they share with the query. ponytail: shared letters rather than an edit + * distance, which would need the whole table scanned to be worth the extra precision. + */ + suspend fun suggest(raw: String, limit: Int = 6): List = withContext(Dispatchers.IO) { + val database = open() ?: return@withContext emptyList() + val word = normalise(raw).takeIf { it.length > 1 } ?: return@withContext emptyList() + + // Widen the prefix until there is something to rank, but never scan the whole table. + val candidates = generateSequence(minOf(3, word.length - 1)) { (it - 1).takeIf { n -> n >= 1 } } + .map { prefixLength -> byPrefix(database, word.take(prefixLength)) } + .firstOrNull { it.size >= 3 } + ?: return@withContext emptyList() + + candidates + .asSequence() + .filter { it.first != word } + .map { (normalised, display) -> display to letterOverlap(word, normalised) } + .filter { it.second > 0.45f } + .sortedByDescending { it.second } + .map { it.first } + .distinct() + .take(limit) + .toList() + } + + /** Index range scan: everything whose normalised form starts with [prefix]. */ + private fun byPrefix(database: SQLiteDatabase, prefix: String): List> = + database.rawQuery( + "SELECT DISTINCT word, display FROM entry WHERE word >= ? AND word < ? LIMIT 400", + arrayOf(prefix, prefix + '\uFFFF'), + ).use { cursor -> + buildList { + while (cursor.moveToNext()) add(cursor.getString(0) to cursor.getString(1)) + } + } + private fun lemmas(database: SQLiteDatabase, form: String): List = database.rawQuery( "SELECT lemma FROM form WHERE form = ? LIMIT 6", @@ -152,6 +193,17 @@ internal fun normalise(text: String, keepZwnj: Boolean = false): String { return (if (keepZwnj) folded else folded.replace(ZWNJ.toString(), "")).trim() } +/** + * How much of two words' letters coincide — the overlapping letters counted against the longer + * word, so مشکل and مشکلها score high while a word that merely starts the same does not. + */ +internal fun letterOverlap(a: String, b: String): Float { + if (a.isEmpty() || b.isEmpty()) return 0f + val remaining = b.toMutableList() + val shared = a.count { remaining.remove(it) } + return shared.toFloat() / maxOf(a.length, b.length) +} + /** The whole word surrounding [index], for turning a tap into something to look up. */ internal fun wordAt(text: String, index: Int): String? { if (text.isEmpty()) return null diff --git a/app/src/main/java/com/ganjoor/android/ui/AboutScreen.kt b/app/src/main/java/com/ganjoor/android/ui/AboutScreen.kt index c96923c..ca12c06 100644 --- a/app/src/main/java/com/ganjoor/android/ui/AboutScreen.kt +++ b/app/src/main/java/com/ganjoor/android/ui/AboutScreen.kt @@ -64,11 +64,17 @@ private val CREDITS = listOf( ), Credit( "Wiktionary", - "Wiktionary contributors — the word definitions, and the inflected-form index that " + - "finds a conjugated verb's dictionary entry", + "Wiktionary contributors — Persian and Urdu definitions, and the inflected-form " + + "index that finds a conjugated verb's dictionary entry", "CC BY-SA 3.0. The bundled dictionary is therefore also CC BY-SA 3.0.", url = "https://en.wiktionary.org", ), + Credit( + "Urdu Wiktionary", + "Urdu Wiktionary contributors — the definitions written in Urdu rather than English", + "CC BY-SA 3.0. The bundled dictionary is therefore also CC BY-SA 3.0.", + url = "https://ur.wiktionary.org", + ), Credit( "Daneshjoo Dictionary", "Layered under Wiktionary for the words it doesn't carry", diff --git a/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt b/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt index 49c88d8..72a19e3 100644 --- a/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt +++ b/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt @@ -7,17 +7,21 @@ import androidx.compose.foundation.layout.navigationBarsPadding import androidx.compose.foundation.layout.padding import androidx.compose.foundation.rememberScrollState import androidx.compose.foundation.verticalScroll +import androidx.compose.foundation.layout.FlowRow import androidx.compose.material3.CircularProgressIndicator import androidx.compose.material3.ExperimentalMaterial3Api import androidx.compose.material3.HorizontalDivider import androidx.compose.material3.MaterialTheme import androidx.compose.material3.ModalBottomSheet +import androidx.compose.material3.SuggestionChip import androidx.compose.material3.Text import androidx.compose.runtime.Composable import androidx.compose.runtime.CompositionLocalProvider import androidx.compose.runtime.getValue import androidx.compose.runtime.mutableStateOf import androidx.compose.runtime.produceState +import androidx.compose.runtime.remember +import androidx.compose.runtime.setValue import androidx.compose.ui.Modifier import androidx.compose.ui.platform.LocalLayoutDirection import androidx.compose.ui.res.stringResource @@ -34,12 +38,39 @@ private fun LeftToRight(content: @Composable () -> Unit) { CompositionLocalProvider(LocalLayoutDirection provides LayoutDirection.Ltr, content = content) } +/** Lays a definition out the way its own script reads. */ +@Composable +private fun InDirectionOf(text: String, content: @Composable () -> Unit) { + val arabicScript = text.count { it in '\u0600'..'\u06FF' } + val latin = text.count { it in 'A'..'Z' || it in 'a'..'z' } + CompositionLocalProvider( + LocalLayoutDirection provides + if (arabicScript > latin) LayoutDirection.Rtl else LayoutDirection.Ltr, + content = content, + ) +} + +/** Which dictionary answered, and in which language pair. */ +private fun sourceLabel(source: String) = when (source) { + "wiktionary-fa" -> R.string.source_wiktionary + "wiktionary-ur" -> R.string.source_wiktionary_ur + "urwiktionary" -> R.string.source_urwiktionary + else -> R.string.source_daneshjoo +} + /** What the dictionary knows about a tapped word. */ @OptIn(ExperimentalMaterial3Api::class) @Composable fun WordSheet(word: String, onDismiss: () -> Unit) { val prefs = LocalSettings.current.value - val definitions by produceState?>(null, word) { value = Dictionary.lookup(word) } + // A suggestion replaces what is being looked up, so the sheet can be followed like a trail. + var current by remember(word) { mutableStateOf(word) } + val definitions by produceState?>(null, current) { + value = Dictionary.lookup(current) + } + val suggestions by produceState(emptyList(), current, definitions) { + value = if (definitions?.isEmpty() == true) Dictionary.suggest(current) else emptyList() + } ModalBottomSheet(onDismissRequest = onDismiss) { Column( @@ -52,7 +83,7 @@ fun WordSheet(word: String, onDismiss: () -> Unit) { ) { // The headword in the reading font, at reading size: it is a line of poetry, after all. Text( - text = word, + text = current, style = readingStyle(prefs.font, prefs.fontSize, prefs.fontWeight.weight), modifier = Modifier.fillMaxWidth(), ) @@ -61,38 +92,52 @@ fun WordSheet(word: String, onDismiss: () -> Unit) { when { definitions == null -> CircularProgressIndicator(Modifier.padding(vertical = 16.dp)) - definitions!!.isEmpty() -> LeftToRight { - Text( - text = stringResource(R.string.no_definition), - style = MaterialTheme.typography.bodyMedium, - color = MaterialTheme.colorScheme.onSurfaceVariant, - modifier = Modifier.fillMaxWidth(), - ) + definitions!!.isEmpty() -> Column( + verticalArrangement = Arrangement.spacedBy(8.dp), + ) { + LeftToRight { + Text( + text = stringResource(R.string.no_definition), + style = MaterialTheme.typography.bodyMedium, + color = MaterialTheme.colorScheme.onSurfaceVariant, + modifier = Modifier.fillMaxWidth(), + ) + } + if (suggestions.isNotEmpty()) { + Text( + text = stringResource(R.string.did_you_mean), + style = MaterialTheme.typography.titleSmall, + color = MaterialTheme.colorScheme.primary, + ) + FlowRow(horizontalArrangement = Arrangement.spacedBy(8.dp)) { + suggestions.forEach { suggestion -> + SuggestionChip( + onClick = { current = suggestion }, + label = { Text(suggestion) }, + ) + } + } + } } else -> definitions!!.forEach { definition -> Column(Modifier.fillMaxWidth()) { // The headword actually matched, which may be the lemma rather than the // word as it appears in the line. Persian, so it stays right-to-left. - if (definition.word != word) { + if (definition.word != current) { Text( text = definition.word, style = MaterialTheme.typography.titleSmall, color = MaterialTheme.colorScheme.primary, ) } - // The definitions are English; right-aligning them reads badly. - LeftToRight { + // Most definitions are English and right-aligning them reads badly; + // the Urdu ones are right-to-left like the rest of the app. + InDirectionOf(definition.gloss) { Column(Modifier.fillMaxWidth()) { Text(definition.gloss, style = MaterialTheme.typography.bodyMedium) Text( - text = stringResource( - if (definition.source == "wiktionary") { - R.string.source_wiktionary - } else { - R.string.source_daneshjoo - } - ), + text = stringResource(sourceLabel(definition.source)), style = MaterialTheme.typography.labelSmall, color = MaterialTheme.colorScheme.onSurfaceVariant, ) diff --git a/app/src/main/res/values-fa/strings.xml b/app/src/main/res/values-fa/strings.xml index b6ccd2b..e78f10a 100644 --- a/app/src/main/res/values-fa/strings.xml +++ b/app/src/main/res/values-fa/strings.xml @@ -65,4 +65,7 @@ ویکی‌واژه — فارسی به انگلیسی (CC BY-SA 3.0) فرهنگ دانشجو — فارسی به انگلیسی لغت‌نامه + ویکی‌واژه — اردو به انگلیسی (CC BY-SA 3.0) + ویکی‌واژهٔ اردو — اردو به اردو (CC BY-SA 3.0) + واژه‌های نزدیک diff --git a/app/src/main/res/values-ur/strings.xml b/app/src/main/res/values-ur/strings.xml index 42d718c..857ccb9 100644 --- a/app/src/main/res/values-ur/strings.xml +++ b/app/src/main/res/values-ur/strings.xml @@ -65,4 +65,7 @@ ویکی لغت — فارسی سے انگریزی (CC BY-SA 3.0) دانشجو لغت — فارسی سے انگریزی لغت نامہ + ویکی لغت — اردو سے انگریزی (CC BY-SA 3.0) + اردو ویکی لغت — اردو سے اردو (CC BY-SA 3.0) + ملتے جلتے الفاظ diff --git a/app/src/main/res/values/strings.xml b/app/src/main/res/values/strings.xml index 4ba5bf9..4466f36 100644 --- a/app/src/main/res/values/strings.xml +++ b/app/src/main/res/values/strings.xml @@ -70,4 +70,7 @@ Wiktionary — Persian to English (CC BY-SA 3.0) Daneshjoo — Persian to English Dictionary + Wiktionary — Urdu to English (CC BY-SA 3.0) + Urdu Wiktionary — Urdu to Urdu (CC BY-SA 3.0) + Similar words diff --git a/app/src/test/java/com/ganjoor/android/DictionaryTest.kt b/app/src/test/java/com/ganjoor/android/DictionaryTest.kt index 7c7855d..8f7f09f 100644 --- a/app/src/test/java/com/ganjoor/android/DictionaryTest.kt +++ b/app/src/test/java/com/ganjoor/android/DictionaryTest.kt @@ -1,6 +1,7 @@ package com.ganjoor.android import com.ganjoor.android.data.affixes +import com.ganjoor.android.data.letterOverlap import com.ganjoor.android.data.normalise import com.ganjoor.android.data.wordAt import org.junit.Assert.assertEquals @@ -106,3 +107,31 @@ class PersianMorphologyTest { assertTrue(affixes("بها").none { it.length < 2 }) } } + +class LetterOverlapTest { + @Test + fun `an identical word overlaps completely`() { + assertEquals(1f, letterOverlap("عشق", "عشق"), 0.001f) + } + + @Test + fun `a suffixed form still scores high against its stem`() { + assertTrue(letterOverlap("مشکل", "مشکلها") > 0.6f) + } + + @Test + fun `sharing only a first letter scores low`() { + assertTrue(letterOverlap("عشق", "عبادتگاه") < 0.4f) + } + + @Test + fun `letters are counted once each, not by presence alone`() { + // ااا against ا shares one letter, not three + assertEquals(1f / 3f, letterOverlap("ااا", "ا"), 0.001f) + } + + @Test + fun `an empty word never matches`() { + assertEquals(0f, letterOverlap("", "عشق"), 0.001f) + } +} diff --git a/licenses/README.md b/licenses/README.md index 1dcd69b..002a7ad 100644 --- a/licenses/README.md +++ b/licenses/README.md @@ -19,6 +19,25 @@ bundled rather than merely linked. Nothing in Ganjoor's repositories restricts AI-assisted use. +## Dictionary + +Four sources, each row in the database tagged with the one it came from so the app can name it +and so any of them can be dropped without rebuilding the others. + +| Source | Direction | Licence | +|---|---|---| +| [Wiktionary](https://en.wiktionary.org) | Persian → English | CC BY-SA 3.0 | +| [Wiktionary](https://en.wiktionary.org) | Urdu → English | CC BY-SA 3.0 | +| [Urdu Wiktionary](https://ur.wiktionary.org) | Urdu → Urdu | CC BY-SA 3.0 | +| [Daneshjoo](https://github.com/0xdolan/Daneshjoo) | Persian → English | Repository states MIT | + +Because three of the four are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**. +Attribution is shown in the app beside every definition. See [`../tools/README.md`](../tools/README.md) +to rebuild it. + +The Daneshjoo entry is worth a caveat: the repository states MIT, but the underlying lexicon is +a published Iranian dictionary, so that relicensing is worth verifying before relying on it. + ## Fonts | Font | Copyright | Licence | File | diff --git a/tools/README.md b/tools/README.md index 43f732f..04e9212 100644 --- a/tools/README.md +++ b/tools/README.md @@ -30,6 +30,20 @@ tables, etymology templates, IPA and descendants. The build keeps the definition form→lemma index (149,589 pairs) and drops the rest. That index is what resolves conjugated verbs: `افتاد → افتادن`, `بگشاید → گشودن`, `دانند → دانستن`. +## Urdu sources + +```sh +curl -L -o ur.jsonl https://kaikki.org/dictionary/Urdu/kaikki.org-dictionary-Urdu.jsonl +curl -L -o urwikt.xml.bz2 \ + https://dumps.wikimedia.org/urwiktionary/latest/urwiktionary-latest-pages-articles.xml.bz2 +bunzip2 -k urwikt.xml.bz2 +``` + +The first gives Urdu headwords glossed in English. The second is Urdu Wiktionary itself, the only +source here whose definitions are written **in Urdu** — thin (around 3,100 usable entries out of +31,000 pages, many being stubs), but for a word it does carry an Urdu reader is better served by +it than by a translation into English. + ## Licences Each row carries its `source`, so attribution stays accurate and either source can be dropped diff --git a/tools/build_dictionary.py b/tools/build_dictionary.py index 2bcddab..06fe21f 100644 --- a/tools/build_dictionary.py +++ b/tools/build_dictionary.py @@ -1,5 +1,11 @@ """Builds the app's dictionary from Wiktionary (CC BY-SA 3.0) and Daneshjoo (MIT). +Three sources, each row tagged so the app can say which one answered and in which language: + wiktionary-fa Persian headwords, English definitions + daneshjoo Persian headwords, English definitions + wiktionary-ur Urdu headwords, English definitions + + Wiktionary's export is 93 MB of linguistic metadata around 0.94 MB of definitions, so this keeps the definitions and the form->lemma index and discards the rest. The `source` column is what keeps the attribution honest and lets either source be dropped later. @@ -36,13 +42,32 @@ for line in open('fa.jsonl', encoding='utf-8'): if gs: pos = e.get('pos') or '' gloss = '; '.join(dict.fromkeys(gs))[:600] - entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary')) + entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa')) for f in e.get('forms', []): t = f.get('form') if t and t != word and not t.startswith('-') and len(t) > 1: forms.add((normalise(t), normalise(word))) -print(f"wiktionary: {len(entries)} entries, {len(forms)} forms") +print(f"wiktionary-fa: {len(entries)} entries, {len(forms)} forms") + +# Urdu shares a great deal of vocabulary with Persian, so these entries answer words the +# Persian sources miss. Headwords are Urdu; the definitions are still English. +n_fa = len(entries) +for line in open('ur.jsonl', encoding='utf-8'): + try: e = json.loads(line) + except Exception: continue + word = e.get('word') + if not word: continue + gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()] + if gs: + pos = e.get('pos') or '' + gloss = '; '.join(dict.fromkeys(gs))[:600] + entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur')) + for f in e.get('forms', []): + t = f.get('form') + if t and t != word and not t.startswith('-') and len(t) > 1: + forms.add((normalise(t), normalise(word))) +print(f"wiktionary-ur: {len(entries) - n_fa} entries, {len(forms)} forms total") tag = re.compile(r'<[^>]+>') n0 = len(entries) @@ -58,6 +83,42 @@ for k, v in MDX('daneshjoo.mdx').items(): entries.append((normalise(word), word, txt[:600], 'daneshjoo')) print(f"daneshjoo : {len(entries) - n0} entries") +# Urdu Wiktionary, the only source here whose definitions are written in Urdu rather than +# English. Thin — a few thousand usable entries, many pages being stubs — but for a word it +# does carry, an Urdu reader is better served by it than by a translation into English. +if os.path.exists('urwikt.xml'): + import html as _html + raw = open('urwikt.xml', encoding='utf-8', errors='ignore').read() + n_ur = len(entries) + for title, ns, body in re.findall( + r'(.*?).*?(\d+).*?]*>(.*?)', raw, re.S + ): + if ns != '0' or not re.match(r'^[\u0600-\u06FF]', title): + continue + t = re.sub(r'\{\{[^}]*\}\}', ' ', body) + t = re.sub(r'\[\[([^\]|]*\|)?([^\]]*)\]\]', r'\2', t) + t = re.sub(r"'{2,}|<[^>]+>", '', _html.unescape(t)) + section = re.search(r'==\s*معانی\s*==(.*?)(?:\n==|\Z)', t, re.S) + lines = (section.group(1) if section else + '\n'.join(l.strip(' #') for l in t.split('\n') if l.strip().startswith('#'))) + kept = [] + for line in lines.split('\n'): + line = re.sub(r'^\d+\.\s*', '', line.strip()) + # ؎ introduces a verse citation, and "ref"/a year starts the source note; the + # definition itself is what comes before either. + if line.startswith('؎') or re.match(r'^\(?\s*\d{3,4}ء', line): + break + line = re.split(r'\bref\b|؎', line)[0].strip() + if not line or not re.search(r'[\u0600-\u06FF]', line): + continue + kept.append(line) + if len(' '.join(kept)) > 220: + break + gloss = ' '.join(kept).strip()[:300] + if len(gloss) > 3: + entries.append((normalise(title), title, gloss, 'urwiktionary')) + print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)") + c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries) c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms)) c.executescript("""