Two Urdu sources join the Persian ones. Wiktionary's Urdu extract gives Urdu headwords glossed in English, and Urdu Wiktionary itself gives definitions written in Urdu — the only source here that does. The latter is thin, about 3,100 usable entries out of 31,000 pages since many are stubs, but for a word it carries an Urdu reader is better served by it than by a translation into English: عشق comes back as شدید جذبۂ محبت، گہری چاہت، محبت، پریم، پیار. Every definition now names the dictionary and its language pair, and lays out in the direction its own script reads, so an Urdu definition is right-aligned beside a left-aligned English one. When nothing matches, the sheet offers near words ranked by how many letters they share with what was looked up, drawn from an index range scan on the leading letters rather than a scan of the whole table. خودکامی, which has no entry, offers خودکامه — the lemma it wants. Measured honestly: the Urdu sources add little coverage over Persian — nine words from the English-glossed extract, two from Urdu Wiktionary, against 1,285 from thirteen poems. They are here because an Urdu reader wants Urdu, not because they widen the net. The suggestion ranking is verified against the built database rather than only on device: خودکامی → خودکامه, شیرازی → شیراز, مشکلها → مشکل. On an API 36 emulator all four sources answer عشق with their labels. Known rough edge: affix stripping across four languages can mislead. ناولها reaches ناول, the Urdu for "novel", which is not what Hafez meant. The matched headword is always shown, so it is visible rather than silent. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
132 lines
5.7 KiB
Python
132 lines
5.7 KiB
Python
"""Builds the app's dictionary from Wiktionary (CC BY-SA 3.0) and Daneshjoo (MIT).
|
||
|
||
Three sources, each row tagged so the app can say which one answered and in which language:
|
||
wiktionary-fa Persian headwords, English definitions
|
||
daneshjoo Persian headwords, English definitions
|
||
wiktionary-ur Urdu headwords, English definitions
|
||
|
||
|
||
Wiktionary's export is 93 MB of linguistic metadata around 0.94 MB of definitions, so this keeps
|
||
the definitions and the form->lemma index and discards the rest. The `source` column is what
|
||
keeps the attribution honest and lets either source be dropped later.
|
||
"""
|
||
import json, re, sqlite3, unicodedata, os, html
|
||
from readmdict import MDX
|
||
|
||
HARAKAT = set(range(0x064B, 0x0653)) | {0x0670, 0x0640} | set(range(0x0610, 0x0616))
|
||
# Arabic letters that Persian writes differently; headwords and poems disagree constantly.
|
||
FOLD = {'ي': 'ی', 'ى': 'ی', 'ك': 'ک', 'ة': 'ه'}
|
||
|
||
def normalise(s: str) -> str:
|
||
d = unicodedata.normalize('NFD', s)
|
||
d = ''.join(c for c in d if ord(c) not in HARAKAT)
|
||
s = unicodedata.normalize('NFC', d)
|
||
return ''.join(FOLD.get(c, c) for c in s).replace('', '').strip()
|
||
|
||
db = 'ganjoor-dictionary.db'
|
||
if os.path.exists(db): os.remove(db)
|
||
c = sqlite3.connect(db)
|
||
c.executescript("""
|
||
CREATE TABLE entry (word TEXT NOT NULL, display TEXT NOT NULL, gloss TEXT NOT NULL, source TEXT NOT NULL);
|
||
CREATE TABLE form (form TEXT NOT NULL, lemma TEXT NOT NULL);
|
||
""")
|
||
|
||
entries, forms = [], set()
|
||
|
||
for line in open('fa.jsonl', encoding='utf-8'):
|
||
try: e = json.loads(line)
|
||
except Exception: continue
|
||
word = e.get('word')
|
||
if not word: continue
|
||
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
|
||
if gs:
|
||
pos = e.get('pos') or ''
|
||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa'))
|
||
for f in e.get('forms', []):
|
||
t = f.get('form')
|
||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||
forms.add((normalise(t), normalise(word)))
|
||
|
||
print(f"wiktionary-fa: {len(entries)} entries, {len(forms)} forms")
|
||
|
||
# Urdu shares a great deal of vocabulary with Persian, so these entries answer words the
|
||
# Persian sources miss. Headwords are Urdu; the definitions are still English.
|
||
n_fa = len(entries)
|
||
for line in open('ur.jsonl', encoding='utf-8'):
|
||
try: e = json.loads(line)
|
||
except Exception: continue
|
||
word = e.get('word')
|
||
if not word: continue
|
||
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
|
||
if gs:
|
||
pos = e.get('pos') or ''
|
||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur'))
|
||
for f in e.get('forms', []):
|
||
t = f.get('form')
|
||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||
forms.add((normalise(t), normalise(word)))
|
||
print(f"wiktionary-ur: {len(entries) - n_fa} entries, {len(forms)} forms total")
|
||
|
||
tag = re.compile(r'<[^>]+>')
|
||
n0 = len(entries)
|
||
for k, v in MDX('daneshjoo.mdx').items():
|
||
word = k.decode('utf-8', 'ignore')
|
||
if not re.match(r'^[-ۿ]', word):
|
||
continue # the en->fa half isn't useful here
|
||
txt = html.unescape(tag.sub(' ', v.decode('utf-8', 'ignore')))
|
||
txt = re.sub(r'\s+', ' ', txt).strip()
|
||
if txt.startswith(word):
|
||
txt = txt[len(word):].strip(' -–—')
|
||
if txt:
|
||
entries.append((normalise(word), word, txt[:600], 'daneshjoo'))
|
||
print(f"daneshjoo : {len(entries) - n0} entries")
|
||
|
||
# Urdu Wiktionary, the only source here whose definitions are written in Urdu rather than
|
||
# English. Thin — a few thousand usable entries, many pages being stubs — but for a word it
|
||
# does carry, an Urdu reader is better served by it than by a translation into English.
|
||
if os.path.exists('urwikt.xml'):
|
||
import html as _html
|
||
raw = open('urwikt.xml', encoding='utf-8', errors='ignore').read()
|
||
n_ur = len(entries)
|
||
for title, ns, body in re.findall(
|
||
r'<title>(.*?)</title>.*?<ns>(\d+)</ns>.*?<text[^>]*>(.*?)</text>', raw, re.S
|
||
):
|
||
if ns != '0' or not re.match(r'^[\u0600-\u06FF]', title):
|
||
continue
|
||
t = re.sub(r'\{\{[^}]*\}\}', ' ', body)
|
||
t = re.sub(r'\[\[([^\]|]*\|)?([^\]]*)\]\]', r'\2', t)
|
||
t = re.sub(r"'{2,}|<[^>]+>", '', _html.unescape(t))
|
||
section = re.search(r'==\s*معانی\s*==(.*?)(?:\n==|\Z)', t, re.S)
|
||
lines = (section.group(1) if section else
|
||
'\n'.join(l.strip(' #') for l in t.split('\n') if l.strip().startswith('#')))
|
||
kept = []
|
||
for line in lines.split('\n'):
|
||
line = re.sub(r'^\d+\.\s*', '', line.strip())
|
||
# ؎ introduces a verse citation, and "ref"/a year starts the source note; the
|
||
# definition itself is what comes before either.
|
||
if line.startswith('؎') or re.match(r'^\(?\s*\d{3,4}ء', line):
|
||
break
|
||
line = re.split(r'\bref\b|؎', line)[0].strip()
|
||
if not line or not re.search(r'[\u0600-\u06FF]', line):
|
||
continue
|
||
kept.append(line)
|
||
if len(' '.join(kept)) > 220:
|
||
break
|
||
gloss = ' '.join(kept).strip()[:300]
|
||
if len(gloss) > 3:
|
||
entries.append((normalise(title), title, gloss, 'urwiktionary'))
|
||
print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)")
|
||
|
||
c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries)
|
||
c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms))
|
||
c.executescript("""
|
||
CREATE INDEX idx_entry_word ON entry(word);
|
||
CREATE INDEX idx_form_form ON form(form);
|
||
""")
|
||
c.commit()
|
||
c.execute("VACUUM")
|
||
c.close()
|
||
print(f"\n{db}: {os.path.getsize(db)/1e6:.1f} MB")
|