Dictionary coverage, pronunciation, and chapter order #1

Merged
anas merged 5 commits from feature/dictionary-coverage-and-chapter-order into main 2026-10-04 16:12:03 +00:00
8 changed files with 63 additions and 2 deletions
Showing only changes of commit b16d020ce0 - Show all commits

Binary file not shown.

View File

@ -17,7 +17,7 @@ data class Definition(val word: String, val gloss: String, val source: String)
*/
object Dictionary {
/** Guards against a copy interrupted half-way leaving an unopenable file behind. */
private const val ASSET_BYTES = 27742208L
private const val ASSET_BYTES = 86663168L
// Not a .gz: the build packager silently gunzips those and drops the extension, which left
// the asset under a different name than the code was opening.
@ -83,9 +83,28 @@ object Dictionary {
}
}
/**
* Ordered the way a reader of this app wants to be answered: a definition written in Urdu
* first, because it needs no translating at all, then the Persian sources, then the ones
* keyed on another language. English is what the rest fall back to, so it comes last by
* coming from the sources that sit last.
*
* Arabic is last outright: its forms index is larger than every other source combined,
* which makes it the likeliest to match something by coincidence.
*/
private fun direct(database: SQLiteDatabase, word: String): List<Definition> =
database.rawQuery(
"SELECT display, gloss, source FROM entry WHERE word = ? LIMIT 12",
"""
SELECT display, gloss, source FROM entry WHERE word = ?
ORDER BY CASE source
WHEN 'urwiktionary' THEN 0
WHEN 'wiktionary-fa' THEN 1
WHEN 'daneshjoo' THEN 2
WHEN 'wiktionary-ur' THEN 3
ELSE 4
END
LIMIT 12
""",
arrayOf(word),
).use { cursor ->
buildList {

View File

@ -55,6 +55,7 @@ private fun sourceLabel(source: String) = when (source) {
"wiktionary-fa" -> R.string.source_wiktionary
"wiktionary-ur" -> R.string.source_wiktionary_ur
"urwiktionary" -> R.string.source_urwiktionary
"wiktionary-ar" -> R.string.source_wiktionary_ar
else -> R.string.source_daneshjoo
}

View File

@ -68,4 +68,5 @@
<string name="source_wiktionary_ur">ویکی‌واژه — اردو به انگلیسی (CC BY-SA 3.0)</string>
<string name="source_urwiktionary">ویکی‌واژهٔ اردو — اردو به اردو (CC BY-SA 3.0)</string>
<string name="did_you_mean">واژه‌های نزدیک</string>
<string name="source_wiktionary_ar">ویکی‌واژه — عربی به انگلیسی (CC BY-SA 3.0)</string>
</resources>

View File

@ -68,4 +68,5 @@
<string name="source_wiktionary_ur">ویکی لغت — اردو سے انگریزی (CC BY-SA 3.0)</string>
<string name="source_urwiktionary">اردو ویکی لغت — اردو سے اردو (CC BY-SA 3.0)</string>
<string name="did_you_mean">ملتے جلتے الفاظ</string>
<string name="source_wiktionary_ar">ویکی لغت — عربی سے انگریزی (CC BY-SA 3.0)</string>
</resources>

View File

@ -73,4 +73,5 @@
<string name="source_wiktionary_ur">Wiktionary — Urdu to English (CC BY-SA 3.0)</string>
<string name="source_urwiktionary">Urdu Wiktionary — Urdu to Urdu (CC BY-SA 3.0)</string>
<string name="did_you_mean">Similar words</string>
<string name="source_wiktionary_ar">Wiktionary — Arabic to English (CC BY-SA 3.0)</string>
</resources>

View File

@ -44,6 +44,25 @@ source here whose definitions are written **in Urdu** — thin (around 3,100 usa
31,000 pages, many being stubs), but for a word it does carry an Urdu reader is better served by
it than by a translation into English.
## Arabic
```sh
curl -L -o ar.jsonl https://kaikki.org/dictionary/Arabic/kaikki.org-dictionary-Arabic.jsonl
```
521 MB, almost all of it the inflection index, and that index is the point: the Arabic quoted
inside Persian verse is conjugated, so السّاقی, الناس, تَلْقَ and تَهْوی only reach a definition
through it. 36,627 entries and 819,608 new form pairs for about 59 MB of database.
## Why there is no Persian-to-Urdu
Wiktionary's Persian entries carry no translations at all — the translation tables live only on
English pages, in a 3.3 GB export. Going Persian to Urdu would mean pivoting through an English
sense, and a sample of that file projects only about 6,900 Persian words with any Urdu
equivalent, most of them modern dictionary vocabulary rather than the language of the poems. The
definitions written in Urdu therefore come from Urdu Wiktionary directly, and are preferred over
the English ones wherever they exist.
## Licences
Each row carries its `source`, so attribution stays accurate and either source can be dropped

View File

@ -69,6 +69,25 @@ for line in open('ur.jsonl', encoding='utf-8'):
forms.add((normalise(t), normalise(word)))
print(f"wiktionary-ur: {len(entries) - n_fa} entries, {len(forms)} forms total")
# Arabic, for the lines classical Persian quotes outright — Hafez opens with one.
if os.path.exists('ar.jsonl'):
n_ar, f_ar = len(entries), len(forms)
for line in open('ar.jsonl', encoding='utf-8'):
try: e = json.loads(line)
except Exception: continue
word = e.get('word')
if not word: continue
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
if gs:
pos = e.get('pos') or ''
gloss = '; '.join(dict.fromkeys(gs))[:600]
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar'))
for f in e.get('forms', []):
t = f.get('form')
if t and t != word and not t.startswith('-') and len(t) > 1:
forms.add((normalise(t), normalise(word)))
print(f"wiktionary-ar: {len(entries) - n_ar} entries, {len(forms) - f_ar} new forms")
tag = re.compile(r'<[^>]+>')
n0 = len(entries)
for k, v in MDX('daneshjoo.mdx').items():