Dictionary coverage, pronunciation, and chapter order #1
Binary file not shown.
@ -17,7 +17,7 @@ data class Definition(val word: String, val gloss: String, val source: String)
|
||||
*/
|
||||
object Dictionary {
|
||||
/** Guards against a copy interrupted half-way leaving an unopenable file behind. */
|
||||
private const val ASSET_BYTES = 27742208L
|
||||
private const val ASSET_BYTES = 86663168L
|
||||
|
||||
// Not a .gz: the build packager silently gunzips those and drops the extension, which left
|
||||
// the asset under a different name than the code was opening.
|
||||
@ -83,9 +83,28 @@ object Dictionary {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Ordered the way a reader of this app wants to be answered: a definition written in Urdu
|
||||
* first, because it needs no translating at all, then the Persian sources, then the ones
|
||||
* keyed on another language. English is what the rest fall back to, so it comes last by
|
||||
* coming from the sources that sit last.
|
||||
*
|
||||
* Arabic is last outright: its forms index is larger than every other source combined,
|
||||
* which makes it the likeliest to match something by coincidence.
|
||||
*/
|
||||
private fun direct(database: SQLiteDatabase, word: String): List<Definition> =
|
||||
database.rawQuery(
|
||||
"SELECT display, gloss, source FROM entry WHERE word = ? LIMIT 12",
|
||||
"""
|
||||
SELECT display, gloss, source FROM entry WHERE word = ?
|
||||
ORDER BY CASE source
|
||||
WHEN 'urwiktionary' THEN 0
|
||||
WHEN 'wiktionary-fa' THEN 1
|
||||
WHEN 'daneshjoo' THEN 2
|
||||
WHEN 'wiktionary-ur' THEN 3
|
||||
ELSE 4
|
||||
END
|
||||
LIMIT 12
|
||||
""",
|
||||
arrayOf(word),
|
||||
).use { cursor ->
|
||||
buildList {
|
||||
|
||||
@ -55,6 +55,7 @@ private fun sourceLabel(source: String) = when (source) {
|
||||
"wiktionary-fa" -> R.string.source_wiktionary
|
||||
"wiktionary-ur" -> R.string.source_wiktionary_ur
|
||||
"urwiktionary" -> R.string.source_urwiktionary
|
||||
"wiktionary-ar" -> R.string.source_wiktionary_ar
|
||||
else -> R.string.source_daneshjoo
|
||||
}
|
||||
|
||||
|
||||
@ -68,4 +68,5 @@
|
||||
<string name="source_wiktionary_ur">ویکیواژه — اردو به انگلیسی (CC BY-SA 3.0)</string>
|
||||
<string name="source_urwiktionary">ویکیواژهٔ اردو — اردو به اردو (CC BY-SA 3.0)</string>
|
||||
<string name="did_you_mean">واژههای نزدیک</string>
|
||||
<string name="source_wiktionary_ar">ویکیواژه — عربی به انگلیسی (CC BY-SA 3.0)</string>
|
||||
</resources>
|
||||
|
||||
@ -68,4 +68,5 @@
|
||||
<string name="source_wiktionary_ur">ویکی لغت — اردو سے انگریزی (CC BY-SA 3.0)</string>
|
||||
<string name="source_urwiktionary">اردو ویکی لغت — اردو سے اردو (CC BY-SA 3.0)</string>
|
||||
<string name="did_you_mean">ملتے جلتے الفاظ</string>
|
||||
<string name="source_wiktionary_ar">ویکی لغت — عربی سے انگریزی (CC BY-SA 3.0)</string>
|
||||
</resources>
|
||||
|
||||
@ -73,4 +73,5 @@
|
||||
<string name="source_wiktionary_ur">Wiktionary — Urdu to English (CC BY-SA 3.0)</string>
|
||||
<string name="source_urwiktionary">Urdu Wiktionary — Urdu to Urdu (CC BY-SA 3.0)</string>
|
||||
<string name="did_you_mean">Similar words</string>
|
||||
<string name="source_wiktionary_ar">Wiktionary — Arabic to English (CC BY-SA 3.0)</string>
|
||||
</resources>
|
||||
|
||||
@ -44,6 +44,25 @@ source here whose definitions are written **in Urdu** — thin (around 3,100 usa
|
||||
31,000 pages, many being stubs), but for a word it does carry an Urdu reader is better served by
|
||||
it than by a translation into English.
|
||||
|
||||
## Arabic
|
||||
|
||||
```sh
|
||||
curl -L -o ar.jsonl https://kaikki.org/dictionary/Arabic/kaikki.org-dictionary-Arabic.jsonl
|
||||
```
|
||||
|
||||
521 MB, almost all of it the inflection index, and that index is the point: the Arabic quoted
|
||||
inside Persian verse is conjugated, so السّاقی, الناس, تَلْقَ and تَهْوی only reach a definition
|
||||
through it. 36,627 entries and 819,608 new form pairs for about 59 MB of database.
|
||||
|
||||
## Why there is no Persian-to-Urdu
|
||||
|
||||
Wiktionary's Persian entries carry no translations at all — the translation tables live only on
|
||||
English pages, in a 3.3 GB export. Going Persian to Urdu would mean pivoting through an English
|
||||
sense, and a sample of that file projects only about 6,900 Persian words with any Urdu
|
||||
equivalent, most of them modern dictionary vocabulary rather than the language of the poems. The
|
||||
definitions written in Urdu therefore come from Urdu Wiktionary directly, and are preferred over
|
||||
the English ones wherever they exist.
|
||||
|
||||
## Licences
|
||||
|
||||
Each row carries its `source`, so attribution stays accurate and either source can be dropped
|
||||
|
||||
@ -69,6 +69,25 @@ for line in open('ur.jsonl', encoding='utf-8'):
|
||||
forms.add((normalise(t), normalise(word)))
|
||||
print(f"wiktionary-ur: {len(entries) - n_fa} entries, {len(forms)} forms total")
|
||||
|
||||
# Arabic, for the lines classical Persian quotes outright — Hafez opens with one.
|
||||
if os.path.exists('ar.jsonl'):
|
||||
n_ar, f_ar = len(entries), len(forms)
|
||||
for line in open('ar.jsonl', encoding='utf-8'):
|
||||
try: e = json.loads(line)
|
||||
except Exception: continue
|
||||
word = e.get('word')
|
||||
if not word: continue
|
||||
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
|
||||
if gs:
|
||||
pos = e.get('pos') or ''
|
||||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar'))
|
||||
for f in e.get('forms', []):
|
||||
t = f.get('form')
|
||||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||||
forms.add((normalise(t), normalise(word)))
|
||||
print(f"wiktionary-ar: {len(entries) - n_ar} entries, {len(forms) - f_ar} new forms")
|
||||
|
||||
tag = re.compile(r'<[^>]+>')
|
||||
n0 = len(entries)
|
||||
for k, v in MDX('daneshjoo.mdx').items():
|
||||
|
||||
Loading…
Reference in New Issue
Block a user