Show how a word is pronounced, Classical Persian first

Wiktionary carries pronunciation for 45,000 of these headwords and the build
was throwing it away. It is now kept: 101,306 entries of IPA, each tagged with
the variety it belongs to, which matters here because a word in a 14th-century
ghazal was not said the way Tehran says it now — عشق is /ˈʔiʃq/ in Classical
Persian and [ʔeʃɢ̥] in Iran today, and the app leads with the former.

The IPA's dots and stress marks are the syllable breakdown. Urdu Wiktionary
adds its own, in Urdu script: the fully vowelled spelling عِشْق and the split
عِش + قوں, both pulled out of its wikitext.

Costs 7.7 MB of database, taking it to 94 MB.

Verified on an API 36 emulator: tapping عشق shows the Classical Persian IPA,
the Urdu syllable split, the vowelled spelling and two modern variants above
the definitions.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Anas RashidandClaude Opus 5 committed 2026-10-04 17:48:12 +02:00
1 parent b16d020ce0
commit 0df736fd89
5 files changed
+92 -4

No files matched your search

+25 -1
View File
@@ -29,9 +29,20 @@ c = sqlite3.connect(db)
c.executescript("""
CREATE TABLE entry (word TEXT NOT NULL, display TEXT NOT NULL, gloss TEXT NOT NULL, source TEXT NOT NULL);
CREATE TABLE form (form TEXT NOT NULL, lemma TEXT NOT NULL);
CREATE TABLE pron (word TEXT NOT NULL, text TEXT NOT NULL, label TEXT NOT NULL, source TEXT NOT NULL);
""")
entries, forms = [], set()
entries, forms, prons = [], set(), set()
def collect_sounds(word, entry, source):
"""IPA with whatever dialect it belongs to. Classical Persian matters most here: this is an
app for poetry written a long time before modern Tehrani vowels."""
for sound in entry.get('sounds') or []:
ipa = (sound.get('ipa') or '').strip()
if not ipa:
continue
tags = [t for t in (sound.get('tags') or []) if t not in ('formal',)]
prons.add((normalise(word), ipa, ' '.join(tags), source))
for line in open('fa.jsonl', encoding='utf-8'):
try: e = json.loads(line)
@@ -43,6 +54,7 @@ for line in open('fa.jsonl', encoding='utf-8'):
pos = e.get('pos') or ''
gloss = '; '.join(dict.fromkeys(gs))[:600]
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa'))
collect_sounds(word, e, 'wiktionary-fa')
for f in e.get('forms', []):
t = f.get('form')
if t and t != word and not t.startswith('-') and len(t) > 1:
@@ -63,6 +75,7 @@ for line in open('ur.jsonl', encoding='utf-8'):
pos = e.get('pos') or ''
gloss = '; '.join(dict.fromkeys(gs))[:600]
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur'))
collect_sounds(word, e, 'wiktionary-ur')
for f in e.get('forms', []):
t = f.get('form')
if t and t != word and not t.startswith('-') and len(t) > 1:
@@ -82,6 +95,7 @@ if os.path.exists('ar.jsonl'):
pos = e.get('pos') or ''
gloss = '; '.join(dict.fromkeys(gs))[:600]
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar'))
collect_sounds(word, e, 'wiktionary-ar')
for f in e.get('forms', []):
t = f.get('form')
if t and t != word and not t.startswith('-') and len(t) > 1:
@@ -136,13 +150,23 @@ if os.path.exists('urwikt.xml'):
gloss = ' '.join(kept).strip()[:300]
if len(gloss) > 3:
entries.append((normalise(title), title, gloss, 'urwiktionary'))
# {عِشْق} is the fully vowelled spelling; "عِش + قوں" is the syllable split
vowelled = re.search(r'\{([\u0600-\u06FF\u064B-\u0652 ]{2,40})\}', t)
if vowelled:
prons.add((normalise(title), vowelled.group(1).strip(), 'اردو', 'urwiktionary'))
syllables = re.search(r'\{([\u0600-\u06FF\u064B-\u0652]+(?: \+ [\u0600-\u06FF\u064B-\u0652]+)+[^}]*)\}', t)
if syllables:
prons.add((normalise(title), syllables.group(1).strip(), 'ہجے', 'urwiktionary'))
print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)")
print(f"pronunciations: {len(prons)}")
c.executemany("INSERT INTO pron VALUES (?,?,?,?)", sorted(prons))
c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries)
c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms))
c.executescript("""
CREATE INDEX idx_entry_word ON entry(word);
CREATE INDEX idx_form_form ON form(form);
CREATE INDEX idx_pron_word ON pron(word);
""")
c.commit()
c.execute("VACUUM")