diff --git a/app/src/main/assets/dictionary.db b/app/src/main/assets/dictionary.db index 88feb42..c6c6d55 100644 Binary files a/app/src/main/assets/dictionary.db and b/app/src/main/assets/dictionary.db differ diff --git a/app/src/main/java/com/ganjoor/android/data/Dictionary.kt b/app/src/main/java/com/ganjoor/android/data/Dictionary.kt index ec3e3ec..a53ff9e 100644 --- a/app/src/main/java/com/ganjoor/android/data/Dictionary.kt +++ b/app/src/main/java/com/ganjoor/android/data/Dictionary.kt @@ -17,7 +17,7 @@ data class Definition(val word: String, val gloss: String, val source: String) */ object Dictionary { /** Guards against a copy interrupted half-way leaving an unopenable file behind. */ - private const val ASSET_BYTES = 27742208L + private const val ASSET_BYTES = 86663168L // Not a .gz: the build packager silently gunzips those and drops the extension, which left // the asset under a different name than the code was opening. @@ -83,9 +83,28 @@ object Dictionary { } } + /** + * Ordered the way a reader of this app wants to be answered: a definition written in Urdu + * first, because it needs no translating at all, then the Persian sources, then the ones + * keyed on another language. English is what the rest fall back to, so it comes last by + * coming from the sources that sit last. + * + * Arabic is last outright: its forms index is larger than every other source combined, + * which makes it the likeliest to match something by coincidence. + */ private fun direct(database: SQLiteDatabase, word: String): List = database.rawQuery( - "SELECT display, gloss, source FROM entry WHERE word = ? LIMIT 12", + """ + SELECT display, gloss, source FROM entry WHERE word = ? + ORDER BY CASE source + WHEN 'urwiktionary' THEN 0 + WHEN 'wiktionary-fa' THEN 1 + WHEN 'daneshjoo' THEN 2 + WHEN 'wiktionary-ur' THEN 3 + ELSE 4 + END + LIMIT 12 + """, arrayOf(word), ).use { cursor -> buildList { diff --git a/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt b/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt index 72a19e3..3815952 100644 --- a/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt +++ b/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt @@ -55,6 +55,7 @@ private fun sourceLabel(source: String) = when (source) { "wiktionary-fa" -> R.string.source_wiktionary "wiktionary-ur" -> R.string.source_wiktionary_ur "urwiktionary" -> R.string.source_urwiktionary + "wiktionary-ar" -> R.string.source_wiktionary_ar else -> R.string.source_daneshjoo } diff --git a/app/src/main/res/values-fa/strings.xml b/app/src/main/res/values-fa/strings.xml index e78f10a..1dcd185 100644 --- a/app/src/main/res/values-fa/strings.xml +++ b/app/src/main/res/values-fa/strings.xml @@ -68,4 +68,5 @@ ویکی‌واژه — اردو به انگلیسی (CC BY-SA 3.0) ویکی‌واژهٔ اردو — اردو به اردو (CC BY-SA 3.0) واژه‌های نزدیک + ویکی‌واژه — عربی به انگلیسی (CC BY-SA 3.0) diff --git a/app/src/main/res/values-ur/strings.xml b/app/src/main/res/values-ur/strings.xml index 857ccb9..d35509b 100644 --- a/app/src/main/res/values-ur/strings.xml +++ b/app/src/main/res/values-ur/strings.xml @@ -68,4 +68,5 @@ ویکی لغت — اردو سے انگریزی (CC BY-SA 3.0) اردو ویکی لغت — اردو سے اردو (CC BY-SA 3.0) ملتے جلتے الفاظ + ویکی لغت — عربی سے انگریزی (CC BY-SA 3.0) diff --git a/app/src/main/res/values/strings.xml b/app/src/main/res/values/strings.xml index 4466f36..9d1d8fd 100644 --- a/app/src/main/res/values/strings.xml +++ b/app/src/main/res/values/strings.xml @@ -73,4 +73,5 @@ Wiktionary — Urdu to English (CC BY-SA 3.0) Urdu Wiktionary — Urdu to Urdu (CC BY-SA 3.0) Similar words + Wiktionary — Arabic to English (CC BY-SA 3.0) diff --git a/tools/README.md b/tools/README.md index 04e9212..8f72f80 100644 --- a/tools/README.md +++ b/tools/README.md @@ -44,6 +44,25 @@ source here whose definitions are written **in Urdu** — thin (around 3,100 usa 31,000 pages, many being stubs), but for a word it does carry an Urdu reader is better served by it than by a translation into English. +## Arabic + +```sh +curl -L -o ar.jsonl https://kaikki.org/dictionary/Arabic/kaikki.org-dictionary-Arabic.jsonl +``` + +521 MB, almost all of it the inflection index, and that index is the point: the Arabic quoted +inside Persian verse is conjugated, so السّاقی, الناس, تَلْقَ and تَهْوی only reach a definition +through it. 36,627 entries and 819,608 new form pairs for about 59 MB of database. + +## Why there is no Persian-to-Urdu + +Wiktionary's Persian entries carry no translations at all — the translation tables live only on +English pages, in a 3.3 GB export. Going Persian to Urdu would mean pivoting through an English +sense, and a sample of that file projects only about 6,900 Persian words with any Urdu +equivalent, most of them modern dictionary vocabulary rather than the language of the poems. The +definitions written in Urdu therefore come from Urdu Wiktionary directly, and are preferred over +the English ones wherever they exist. + ## Licences Each row carries its `source`, so attribution stays accurate and either source can be dropped diff --git a/tools/build_dictionary.py b/tools/build_dictionary.py index 06fe21f..57f7c7e 100644 --- a/tools/build_dictionary.py +++ b/tools/build_dictionary.py @@ -69,6 +69,25 @@ for line in open('ur.jsonl', encoding='utf-8'): forms.add((normalise(t), normalise(word))) print(f"wiktionary-ur: {len(entries) - n_fa} entries, {len(forms)} forms total") +# Arabic, for the lines classical Persian quotes outright — Hafez opens with one. +if os.path.exists('ar.jsonl'): + n_ar, f_ar = len(entries), len(forms) + for line in open('ar.jsonl', encoding='utf-8'): + try: e = json.loads(line) + except Exception: continue + word = e.get('word') + if not word: continue + gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()] + if gs: + pos = e.get('pos') or '' + gloss = '; '.join(dict.fromkeys(gs))[:600] + entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar')) + for f in e.get('forms', []): + t = f.get('form') + if t and t != word and not t.startswith('-') and len(t) > 1: + forms.add((normalise(t), normalise(word))) + print(f"wiktionary-ar: {len(entries) - n_ar} entries, {len(forms) - f_ar} new forms") + tag = re.compile(r'<[^>]+>') n0 = len(entries) for k, v in MDX('daneshjoo.mdx').items():