"""Builds the app's dictionary from Wiktionary (CC BY-SA 3.0) and Daneshjoo (MIT). Three sources, each row tagged so the app can say which one answered and in which language: wiktionary-fa Persian headwords, English definitions daneshjoo Persian headwords, English definitions wiktionary-ur Urdu headwords, English definitions Wiktionary's export is 93 MB of linguistic metadata around 0.94 MB of definitions, so this keeps the definitions and the form->lemma index and discards the rest. The `source` column is what keeps the attribution honest and lets either source be dropped later. """ import json, re, sqlite3, unicodedata, os, html from readmdict import MDX HARAKAT = set(range(0x064B, 0x0653)) | {0x0670, 0x0640} | set(range(0x0610, 0x0616)) # Arabic letters that Persian writes differently; headwords and poems disagree constantly. FOLD = {'ي': 'ی', 'ى': 'ی', 'ك': 'ک', 'ة': 'ه'} def normalise(s: str) -> str: d = unicodedata.normalize('NFD', s) d = ''.join(c for c in d if ord(c) not in HARAKAT) s = unicodedata.normalize('NFC', d) return ''.join(FOLD.get(c, c) for c in s).replace('', '').strip() db = 'ganjoor-dictionary.db' if os.path.exists(db): os.remove(db) c = sqlite3.connect(db) c.executescript(""" CREATE TABLE entry (word TEXT NOT NULL, display TEXT NOT NULL, gloss TEXT NOT NULL, source TEXT NOT NULL); CREATE TABLE form (form TEXT NOT NULL, lemma TEXT NOT NULL); """) entries, forms = [], set() for line in open('fa.jsonl', encoding='utf-8'): try: e = json.loads(line) except Exception: continue word = e.get('word') if not word: continue gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()] if gs: pos = e.get('pos') or '' gloss = '; '.join(dict.fromkeys(gs))[:600] entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa')) for f in e.get('forms', []): t = f.get('form') if t and t != word and not t.startswith('-') and len(t) > 1: forms.add((normalise(t), normalise(word))) print(f"wiktionary-fa: {len(entries)} entries, {len(forms)} forms") # Urdu shares a great deal of vocabulary with Persian, so these entries answer words the # Persian sources miss. Headwords are Urdu; the definitions are still English. n_fa = len(entries) for line in open('ur.jsonl', encoding='utf-8'): try: e = json.loads(line) except Exception: continue word = e.get('word') if not word: continue gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()] if gs: pos = e.get('pos') or '' gloss = '; '.join(dict.fromkeys(gs))[:600] entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur')) for f in e.get('forms', []): t = f.get('form') if t and t != word and not t.startswith('-') and len(t) > 1: forms.add((normalise(t), normalise(word))) print(f"wiktionary-ur: {len(entries) - n_fa} entries, {len(forms)} forms total") # Arabic, for the lines classical Persian quotes outright — Hafez opens with one. if os.path.exists('ar.jsonl'): n_ar, f_ar = len(entries), len(forms) for line in open('ar.jsonl', encoding='utf-8'): try: e = json.loads(line) except Exception: continue word = e.get('word') if not word: continue gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()] if gs: pos = e.get('pos') or '' gloss = '; '.join(dict.fromkeys(gs))[:600] entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar')) for f in e.get('forms', []): t = f.get('form') if t and t != word and not t.startswith('-') and len(t) > 1: forms.add((normalise(t), normalise(word))) print(f"wiktionary-ar: {len(entries) - n_ar} entries, {len(forms) - f_ar} new forms") tag = re.compile(r'<[^>]+>') n0 = len(entries) for k, v in MDX('daneshjoo.mdx').items(): word = k.decode('utf-8', 'ignore') if not re.match(r'^[-ۿ]', word): continue # the en->fa half isn't useful here txt = html.unescape(tag.sub(' ', v.decode('utf-8', 'ignore'))) txt = re.sub(r'\s+', ' ', txt).strip() if txt.startswith(word): txt = txt[len(word):].strip(' -–—') if txt: entries.append((normalise(word), word, txt[:600], 'daneshjoo')) print(f"daneshjoo : {len(entries) - n0} entries") # Urdu Wiktionary, the only source here whose definitions are written in Urdu rather than # English. Thin — a few thousand usable entries, many pages being stubs — but for a word it # does carry, an Urdu reader is better served by it than by a translation into English. if os.path.exists('urwikt.xml'): import html as _html raw = open('urwikt.xml', encoding='utf-8', errors='ignore').read() n_ur = len(entries) for title, ns, body in re.findall( r'