Wiktionary carries pronunciation for 45,000 of these headwords and the build was throwing it away. It is now kept: 101,306 entries of IPA, each tagged with the variety it belongs to, which matters here because a word in a 14th-century ghazal was not said the way Tehran says it now — عشق is /ˈʔiʃq/ in Classical Persian and [ʔeʃɢ̥] in Iran today, and the app leads with the former. The IPA's dots and stress marks are the syllable breakdown. Urdu Wiktionary adds its own, in Urdu script: the fully vowelled spelling عِشْق and the split عِش + قوں, both pulled out of its wikitext. Costs 7.7 MB of database, taking it to 94 MB. Verified on an API 36 emulator: tapping عشق shows the Classical Persian IPA, the Urdu syllable split, the vowelled spelling and two modern variants above the definitions. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
175 lines
8.0 KiB
Python
175 lines
8.0 KiB
Python
"""Builds the app's dictionary from Wiktionary (CC BY-SA 3.0) and Daneshjoo (MIT).
|
||
|
||
Three sources, each row tagged so the app can say which one answered and in which language:
|
||
wiktionary-fa Persian headwords, English definitions
|
||
daneshjoo Persian headwords, English definitions
|
||
wiktionary-ur Urdu headwords, English definitions
|
||
|
||
|
||
Wiktionary's export is 93 MB of linguistic metadata around 0.94 MB of definitions, so this keeps
|
||
the definitions and the form->lemma index and discards the rest. The `source` column is what
|
||
keeps the attribution honest and lets either source be dropped later.
|
||
"""
|
||
import json, re, sqlite3, unicodedata, os, html
|
||
from readmdict import MDX
|
||
|
||
HARAKAT = set(range(0x064B, 0x0653)) | {0x0670, 0x0640} | set(range(0x0610, 0x0616))
|
||
# Arabic letters that Persian writes differently; headwords and poems disagree constantly.
|
||
FOLD = {'ي': 'ی', 'ى': 'ی', 'ك': 'ک', 'ة': 'ه'}
|
||
|
||
def normalise(s: str) -> str:
|
||
d = unicodedata.normalize('NFD', s)
|
||
d = ''.join(c for c in d if ord(c) not in HARAKAT)
|
||
s = unicodedata.normalize('NFC', d)
|
||
return ''.join(FOLD.get(c, c) for c in s).replace('', '').strip()
|
||
|
||
db = 'ganjoor-dictionary.db'
|
||
if os.path.exists(db): os.remove(db)
|
||
c = sqlite3.connect(db)
|
||
c.executescript("""
|
||
CREATE TABLE entry (word TEXT NOT NULL, display TEXT NOT NULL, gloss TEXT NOT NULL, source TEXT NOT NULL);
|
||
CREATE TABLE form (form TEXT NOT NULL, lemma TEXT NOT NULL);
|
||
CREATE TABLE pron (word TEXT NOT NULL, text TEXT NOT NULL, label TEXT NOT NULL, source TEXT NOT NULL);
|
||
""")
|
||
|
||
entries, forms, prons = [], set(), set()
|
||
|
||
def collect_sounds(word, entry, source):
|
||
"""IPA with whatever dialect it belongs to. Classical Persian matters most here: this is an
|
||
app for poetry written a long time before modern Tehrani vowels."""
|
||
for sound in entry.get('sounds') or []:
|
||
ipa = (sound.get('ipa') or '').strip()
|
||
if not ipa:
|
||
continue
|
||
tags = [t for t in (sound.get('tags') or []) if t not in ('formal',)]
|
||
prons.add((normalise(word), ipa, ' '.join(tags), source))
|
||
|
||
for line in open('fa.jsonl', encoding='utf-8'):
|
||
try: e = json.loads(line)
|
||
except Exception: continue
|
||
word = e.get('word')
|
||
if not word: continue
|
||
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
|
||
if gs:
|
||
pos = e.get('pos') or ''
|
||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa'))
|
||
collect_sounds(word, e, 'wiktionary-fa')
|
||
for f in e.get('forms', []):
|
||
t = f.get('form')
|
||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||
forms.add((normalise(t), normalise(word)))
|
||
|
||
print(f"wiktionary-fa: {len(entries)} entries, {len(forms)} forms")
|
||
|
||
# Urdu shares a great deal of vocabulary with Persian, so these entries answer words the
|
||
# Persian sources miss. Headwords are Urdu; the definitions are still English.
|
||
n_fa = len(entries)
|
||
for line in open('ur.jsonl', encoding='utf-8'):
|
||
try: e = json.loads(line)
|
||
except Exception: continue
|
||
word = e.get('word')
|
||
if not word: continue
|
||
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
|
||
if gs:
|
||
pos = e.get('pos') or ''
|
||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur'))
|
||
collect_sounds(word, e, 'wiktionary-ur')
|
||
for f in e.get('forms', []):
|
||
t = f.get('form')
|
||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||
forms.add((normalise(t), normalise(word)))
|
||
print(f"wiktionary-ur: {len(entries) - n_fa} entries, {len(forms)} forms total")
|
||
|
||
# Arabic, for the lines classical Persian quotes outright — Hafez opens with one.
|
||
if os.path.exists('ar.jsonl'):
|
||
n_ar, f_ar = len(entries), len(forms)
|
||
for line in open('ar.jsonl', encoding='utf-8'):
|
||
try: e = json.loads(line)
|
||
except Exception: continue
|
||
word = e.get('word')
|
||
if not word: continue
|
||
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
|
||
if gs:
|
||
pos = e.get('pos') or ''
|
||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar'))
|
||
collect_sounds(word, e, 'wiktionary-ar')
|
||
for f in e.get('forms', []):
|
||
t = f.get('form')
|
||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||
forms.add((normalise(t), normalise(word)))
|
||
print(f"wiktionary-ar: {len(entries) - n_ar} entries, {len(forms) - f_ar} new forms")
|
||
|
||
tag = re.compile(r'<[^>]+>')
|
||
n0 = len(entries)
|
||
for k, v in MDX('daneshjoo.mdx').items():
|
||
word = k.decode('utf-8', 'ignore')
|
||
if not re.match(r'^[-ۿ]', word):
|
||
continue # the en->fa half isn't useful here
|
||
txt = html.unescape(tag.sub(' ', v.decode('utf-8', 'ignore')))
|
||
txt = re.sub(r'\s+', ' ', txt).strip()
|
||
if txt.startswith(word):
|
||
txt = txt[len(word):].strip(' -–—')
|
||
if txt:
|
||
entries.append((normalise(word), word, txt[:600], 'daneshjoo'))
|
||
print(f"daneshjoo : {len(entries) - n0} entries")
|
||
|
||
# Urdu Wiktionary, the only source here whose definitions are written in Urdu rather than
|
||
# English. Thin — a few thousand usable entries, many pages being stubs — but for a word it
|
||
# does carry, an Urdu reader is better served by it than by a translation into English.
|
||
if os.path.exists('urwikt.xml'):
|
||
import html as _html
|
||
raw = open('urwikt.xml', encoding='utf-8', errors='ignore').read()
|
||
n_ur = len(entries)
|
||
for title, ns, body in re.findall(
|
||
r'<title>(.*?)</title>.*?<ns>(\d+)</ns>.*?<text[^>]*>(.*?)</text>', raw, re.S
|
||
):
|
||
if ns != '0' or not re.match(r'^[\u0600-\u06FF]', title):
|
||
continue
|
||
t = re.sub(r'\{\{[^}]*\}\}', ' ', body)
|
||
t = re.sub(r'\[\[([^\]|]*\|)?([^\]]*)\]\]', r'\2', t)
|
||
t = re.sub(r"'{2,}|<[^>]+>", '', _html.unescape(t))
|
||
section = re.search(r'==\s*معانی\s*==(.*?)(?:\n==|\Z)', t, re.S)
|
||
lines = (section.group(1) if section else
|
||
'\n'.join(l.strip(' #') for l in t.split('\n') if l.strip().startswith('#')))
|
||
kept = []
|
||
for line in lines.split('\n'):
|
||
line = re.sub(r'^\d+\.\s*', '', line.strip())
|
||
# ؎ introduces a verse citation, and "ref"/a year starts the source note; the
|
||
# definition itself is what comes before either.
|
||
if line.startswith('؎') or re.match(r'^\(?\s*\d{3,4}ء', line):
|
||
break
|
||
line = re.split(r'\bref\b|؎', line)[0].strip()
|
||
if not line or not re.search(r'[\u0600-\u06FF]', line):
|
||
continue
|
||
kept.append(line)
|
||
if len(' '.join(kept)) > 220:
|
||
break
|
||
gloss = ' '.join(kept).strip()[:300]
|
||
if len(gloss) > 3:
|
||
entries.append((normalise(title), title, gloss, 'urwiktionary'))
|
||
# {عِشْق} is the fully vowelled spelling; "عِش + قوں" is the syllable split
|
||
vowelled = re.search(r'\{([\u0600-\u06FF\u064B-\u0652 ]{2,40})\}', t)
|
||
if vowelled:
|
||
prons.add((normalise(title), vowelled.group(1).strip(), 'اردو', 'urwiktionary'))
|
||
syllables = re.search(r'\{([\u0600-\u06FF\u064B-\u0652]+(?: \+ [\u0600-\u06FF\u064B-\u0652]+)+[^}]*)\}', t)
|
||
if syllables:
|
||
prons.add((normalise(title), syllables.group(1).strip(), 'ہجے', 'urwiktionary'))
|
||
print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)")
|
||
|
||
print(f"pronunciations: {len(prons)}")
|
||
c.executemany("INSERT INTO pron VALUES (?,?,?,?)", sorted(prons))
|
||
c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries)
|
||
c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms))
|
||
c.executescript("""
|
||
CREATE INDEX idx_entry_word ON entry(word);
|
||
CREATE INDEX idx_form_form ON form(form);
|
||
CREATE INDEX idx_pron_word ON pron(word);
|
||
""")
|
||
c.commit()
|
||
c.execute("VACUUM")
|
||
c.close()
|
||
print(f"\n{db}: {os.path.getsize(db)/1e6:.1f} MB")
|