Classical Persian quotes Arabic outright — Hafez opens with a whole hemistich of it — and none of it resolved before. The full Arabic export brings 36,627 entries and 819,608 new form pairs, and the forms are the point: السّاقی, الناس, تَلْقَ and تَهْوی are all conjugated or carry the article, so they only reach a definition through that index. Nine of the ten Arabic words in the sample now answer where none did. The cost is real and worth stating: the database goes from 28 MB to 87 MB and the release APK from 10.6 MB to 29.2 MB, with another 87 MB unpacked on first run, so about 116 MB installed. Results are now ordered the way a reader of this app wants them: the Urdu definition first, because it needs no translating, then Persian, then sources keyed on another language, with English arriving only through whatever is left. Arabic sorts last outright — its index is larger than every other source combined, which makes it the likeliest to match by coincidence. No Persian-to-Urdu dictionary. Wiktionary's Persian entries carry no translations at all; the tables live only on English pages, in a 3.3 GB export, so it would mean pivoting through an English sense. A sample of that file projects about 6,900 Persian words with any Urdu equivalent, mostly modern dictionary vocabulary rather than the language of the poems — not worth the pivot. tools/README.md records why, so the question doesn't get reopened from scratch. Verified on an API 36 emulator: عشق returns all five sources with the Urdu definition at the top. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
151 lines
6.7 KiB
Python
151 lines
6.7 KiB
Python
"""Builds the app's dictionary from Wiktionary (CC BY-SA 3.0) and Daneshjoo (MIT).
|
||
|
||
Three sources, each row tagged so the app can say which one answered and in which language:
|
||
wiktionary-fa Persian headwords, English definitions
|
||
daneshjoo Persian headwords, English definitions
|
||
wiktionary-ur Urdu headwords, English definitions
|
||
|
||
|
||
Wiktionary's export is 93 MB of linguistic metadata around 0.94 MB of definitions, so this keeps
|
||
the definitions and the form->lemma index and discards the rest. The `source` column is what
|
||
keeps the attribution honest and lets either source be dropped later.
|
||
"""
|
||
import json, re, sqlite3, unicodedata, os, html
|
||
from readmdict import MDX
|
||
|
||
HARAKAT = set(range(0x064B, 0x0653)) | {0x0670, 0x0640} | set(range(0x0610, 0x0616))
|
||
# Arabic letters that Persian writes differently; headwords and poems disagree constantly.
|
||
FOLD = {'ي': 'ی', 'ى': 'ی', 'ك': 'ک', 'ة': 'ه'}
|
||
|
||
def normalise(s: str) -> str:
|
||
d = unicodedata.normalize('NFD', s)
|
||
d = ''.join(c for c in d if ord(c) not in HARAKAT)
|
||
s = unicodedata.normalize('NFC', d)
|
||
return ''.join(FOLD.get(c, c) for c in s).replace('', '').strip()
|
||
|
||
db = 'ganjoor-dictionary.db'
|
||
if os.path.exists(db): os.remove(db)
|
||
c = sqlite3.connect(db)
|
||
c.executescript("""
|
||
CREATE TABLE entry (word TEXT NOT NULL, display TEXT NOT NULL, gloss TEXT NOT NULL, source TEXT NOT NULL);
|
||
CREATE TABLE form (form TEXT NOT NULL, lemma TEXT NOT NULL);
|
||
""")
|
||
|
||
entries, forms = [], set()
|
||
|
||
for line in open('fa.jsonl', encoding='utf-8'):
|
||
try: e = json.loads(line)
|
||
except Exception: continue
|
||
word = e.get('word')
|
||
if not word: continue
|
||
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
|
||
if gs:
|
||
pos = e.get('pos') or ''
|
||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa'))
|
||
for f in e.get('forms', []):
|
||
t = f.get('form')
|
||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||
forms.add((normalise(t), normalise(word)))
|
||
|
||
print(f"wiktionary-fa: {len(entries)} entries, {len(forms)} forms")
|
||
|
||
# Urdu shares a great deal of vocabulary with Persian, so these entries answer words the
|
||
# Persian sources miss. Headwords are Urdu; the definitions are still English.
|
||
n_fa = len(entries)
|
||
for line in open('ur.jsonl', encoding='utf-8'):
|
||
try: e = json.loads(line)
|
||
except Exception: continue
|
||
word = e.get('word')
|
||
if not word: continue
|
||
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
|
||
if gs:
|
||
pos = e.get('pos') or ''
|
||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur'))
|
||
for f in e.get('forms', []):
|
||
t = f.get('form')
|
||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||
forms.add((normalise(t), normalise(word)))
|
||
print(f"wiktionary-ur: {len(entries) - n_fa} entries, {len(forms)} forms total")
|
||
|
||
# Arabic, for the lines classical Persian quotes outright — Hafez opens with one.
|
||
if os.path.exists('ar.jsonl'):
|
||
n_ar, f_ar = len(entries), len(forms)
|
||
for line in open('ar.jsonl', encoding='utf-8'):
|
||
try: e = json.loads(line)
|
||
except Exception: continue
|
||
word = e.get('word')
|
||
if not word: continue
|
||
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
|
||
if gs:
|
||
pos = e.get('pos') or ''
|
||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar'))
|
||
for f in e.get('forms', []):
|
||
t = f.get('form')
|
||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||
forms.add((normalise(t), normalise(word)))
|
||
print(f"wiktionary-ar: {len(entries) - n_ar} entries, {len(forms) - f_ar} new forms")
|
||
|
||
tag = re.compile(r'<[^>]+>')
|
||
n0 = len(entries)
|
||
for k, v in MDX('daneshjoo.mdx').items():
|
||
word = k.decode('utf-8', 'ignore')
|
||
if not re.match(r'^[-ۿ]', word):
|
||
continue # the en->fa half isn't useful here
|
||
txt = html.unescape(tag.sub(' ', v.decode('utf-8', 'ignore')))
|
||
txt = re.sub(r'\s+', ' ', txt).strip()
|
||
if txt.startswith(word):
|
||
txt = txt[len(word):].strip(' -–—')
|
||
if txt:
|
||
entries.append((normalise(word), word, txt[:600], 'daneshjoo'))
|
||
print(f"daneshjoo : {len(entries) - n0} entries")
|
||
|
||
# Urdu Wiktionary, the only source here whose definitions are written in Urdu rather than
|
||
# English. Thin — a few thousand usable entries, many pages being stubs — but for a word it
|
||
# does carry, an Urdu reader is better served by it than by a translation into English.
|
||
if os.path.exists('urwikt.xml'):
|
||
import html as _html
|
||
raw = open('urwikt.xml', encoding='utf-8', errors='ignore').read()
|
||
n_ur = len(entries)
|
||
for title, ns, body in re.findall(
|
||
r'<title>(.*?)</title>.*?<ns>(\d+)</ns>.*?<text[^>]*>(.*?)</text>', raw, re.S
|
||
):
|
||
if ns != '0' or not re.match(r'^[\u0600-\u06FF]', title):
|
||
continue
|
||
t = re.sub(r'\{\{[^}]*\}\}', ' ', body)
|
||
t = re.sub(r'\[\[([^\]|]*\|)?([^\]]*)\]\]', r'\2', t)
|
||
t = re.sub(r"'{2,}|<[^>]+>", '', _html.unescape(t))
|
||
section = re.search(r'==\s*معانی\s*==(.*?)(?:\n==|\Z)', t, re.S)
|
||
lines = (section.group(1) if section else
|
||
'\n'.join(l.strip(' #') for l in t.split('\n') if l.strip().startswith('#')))
|
||
kept = []
|
||
for line in lines.split('\n'):
|
||
line = re.sub(r'^\d+\.\s*', '', line.strip())
|
||
# ؎ introduces a verse citation, and "ref"/a year starts the source note; the
|
||
# definition itself is what comes before either.
|
||
if line.startswith('؎') or re.match(r'^\(?\s*\d{3,4}ء', line):
|
||
break
|
||
line = re.split(r'\bref\b|؎', line)[0].strip()
|
||
if not line or not re.search(r'[\u0600-\u06FF]', line):
|
||
continue
|
||
kept.append(line)
|
||
if len(' '.join(kept)) > 220:
|
||
break
|
||
gloss = ' '.join(kept).strip()[:300]
|
||
if len(gloss) > 3:
|
||
entries.append((normalise(title), title, gloss, 'urwiktionary'))
|
||
print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)")
|
||
|
||
c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries)
|
||
c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms))
|
||
c.executescript("""
|
||
CREATE INDEX idx_entry_word ON entry(word);
|
||
CREATE INDEX idx_form_form ON form(form);
|
||
""")
|
||
c.commit()
|
||
c.execute("VACUUM")
|
||
c.close()
|
||
print(f"\n{db}: {os.path.getsize(db)/1e6:.1f} MB")
|