// Word dictionary for the reading sidebar: the full Wiktionary data for Urdu, Persian and Arabic, in PostgreSQL.
// source 'en': en.wiktionary entries (English meanings, IPA, transliteration, audio, etymology, synonyms),
// from the kaikki.org Wiktextract extracts (weekly)
// source 'own': each language's own Wiktionary (ur., fa., ar.wiktionary): meanings in that language,
// from the Wikimedia dumps (twice a month) plus recent changes (daily)
// Urdu equivalents for words with no Urdu meaning are pivoted through English: ur_glosses maps English words to
// Urdu words, from the English glosses of en.wiktionary's Urdu entries and ur.wiktionary's English entries. Data: CC BY-SA, Wiktionary contributors.
import { spawn } from 'node:child_process';
import { createInterface } from 'node:readline';
import { Readable } from 'node:stream';
import { pool } from './db.ts';
export const LANGS = ['ur', 'fa', 'ar'] as const;
const NAME = { ur: 'Urdu', fa: 'Persian', ar: 'Arabic' } as const;
const UA = 'Divan/2 (https://github.com/anas-rashid/divan)';
const KAIKKI = (l: string) => `https://kaikki.org/dictionary/${NAME[l as 'ur']}/kaikki.org-dictionary-${NAME[l as 'ur']}.jsonl`;
const DUMP = (l: string) => `https://dumps.wikimedia.org/${l}wiktionary/latest/${l}wiktionary-latest-pages-articles.xml.bz2`;
const MARKS = /[ؐ-ًؚ-ٰٟۖ-ۭـ]/g;
// punctuation, dashes and spaces: dropped from words (بے ثبوت = بےثبوت, دل، = دل)
export const PUNCT = /[\s\-‐-―_.,،۔؛؟!?:;"'«»()[\]{}/\\*]+/g;
// spelling-insensitive key shared by Urdu, Persian and Arabic: no diacritics/tatweel/punctuation/spaces, one heh,
// one yeh, one kaf, plain alef, noon ghunna as noon (ناداں = نادان, قسمت = قِسْمَت, نگاہ = نگاه, معنی = معنى).
// ے and ھ stay distinct.
export const key = (w: string) =>
w.normalize('NFC').replace(MARKS, '').replace(PUNCT, '')
.replace(/[ہۂۃةه]/g, 'ه').replace(/[يىی]/g, 'ی')
.replace(/ك/g, 'ک').replace(/[أإٱ]/g, 'ا').replace(/ں/g, 'ن');
const unlink = (wt: string) => wt.replace(/\[\[(?:[^\]|]*\|)?([^\]]*)\]\]/g, '$1').replace(/'''?/g, '').trim();
const upload = (u?: string) => (u && u.startsWith('https://upload.wikimedia.org/') ? u : null);
// ---- parsing ----
// a kaikki.org (Wiktextract) line -> the fields the sidebar shows
export function kaikkiEntry(d: any) {
const senses = (d.senses ?? []).filter((s: any) => s.glosses?.length);
const sounds = d.sounds ?? [];
return {
pos: d.pos as string,
glosses: senses.map((s: any) => s.glosses.join('; ')).slice(0, 6) as string[],
formOf: senses.length > 0 && senses.every((s: any) => s.form_of || s.tags?.includes('form-of')),
ipa: [...new Set(sounds.map((s: any) => s.ipa).filter(Boolean))].slice(0, 2) as string[],
audio: sounds.map((s: any) => upload(s.mp3_url) ?? upload(s.ogg_url)).find(Boolean) ?? null,
tr: d.forms?.find((f: any) => f.tags?.includes('romanization'))?.form ?? null,
form: d.forms?.find((f: any) => f.tags?.includes('canonical'))?.form ?? null, // with short vowels: مُلْک
ety: d.etymology_text ? etymology(String(d.etymology_text)) : null,
synonyms: (d.synonyms ?? []).map((s: any) => s.word).filter(Boolean).slice(0, 8) as string[],
};
}
// etymology text without Wiktionary's "Etymology tree …" diagram summary; cut at 300 characters, never
// leaving half a surrogate pair
const etymology = (t: string) =>
(t.startsWith('Etymology tree') ? t.slice(Math.max(0, t.search(/\b(Borrowed|Inherited|From|Learned|Derived|Semi-learned|Calque|Compound)\b/))) : t)
.slice(0, 300).toWellFormed();
// English glosses of an Urdu entry as pivot keys: "fate, destiny" -> fate, destiny (short, lowercase)
export const glossKeys = (glosses: string[]) => [...new Set(glosses.flatMap((g) => g.replace(/\([^)]*\)/g, '').split(/[,;]/))
.map((s) => s.trim().toLowerCase().replace(/^(to|a|an|the) /, '')).filter((s) => /^[a-z][a-z' -]{1,30}$/.test(s) && s.split(' ').length <= 3 && !FILLER.has(s)))];
const FILLER = new Set(['especially', 'usually', 'often', 'also', 'etc', 'something', 'someone', 'chiefly', 'mainly', 'figuratively', 'literally', 'rare', 'archaic', 'obsolete', 'dated', 'poetic', 'colloquial', 'informal', 'formal', 'slang', 'by extension']);
// ur.wiktionary Urdu entries: numbered lines under ==معانی==, else "# " lines, else the opening prose;
// origin from "(عربی)" on the first line
export function urduEntry(wt: string) {
const m = wt.split(/==\s*معانی\s*==/)[1]?.split(/\n==/)[0] ?? '';
let defs = m.split('\n').map((l) => l.match(/^\s*(?:\d+[.)-]|#)\s*(.+)/)?.[1]).filter(Boolean).map((l) => unlink(l!));
if (!defs.length) defs = definitions(wt, 'ur').defs;
if (!defs.length) {
const prose = wt.split('\n').find((l) => l.trim().startsWith("'''") && l.length > 20);
if (prose) defs = [unlink(stripTemplates(prose)).slice(0, 300)];
}
const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null;
const english = wt.match(/^\s*انگریزی\s*:\s*([A-Za-z].+)$/m)?.[1].trim() ?? null; // "== تراجم ==" pages
return { defs: defs.slice(0, 6), origin, links: [] as string[], ...(english && { english }) };
}
// ur.wiktionary English entries ("north": "# [[شمالی]]۔"): the Urdu words, for the pivot through English
export const isEnglish = (title: string) => /^[A-Za-z][A-Za-z' -]*$/.test(title);
export function englishToUrdu(wt: string) {
const lines = wt.split('\n').filter((l) => /^#(?![:*])/.test(l)).join(' ');
return [...new Set([...lines.matchAll(/\[\[(?:[^\]|]*\|)?([^\]]+)\]\]/g)].map((m) => m[1].trim())
.filter((w) => /^[\p{Script=Arabic}\s]+$/u.test(w)))].slice(0, 10);
}
const stripTemplates = (t: string) => {
while (/\{\{[^{}]*\}\}/.test(t)) t = t.replace(/\{\{[^{}]*\}\}/g, '');
return t;
};
// fa./ar.wiktionary: "# definition" lines outside etymology/translation sections; on ar.wiktionary only the
// Arabic section, and undiacritised pages that just point to entries ("* [[عِشْق]]") give those links
export function definitions(wt: string, code: string) {
if (code === 'ar' && wt.includes('{{اللغة|')) wt = wt.split(/==\s*\{\{اللغة\|/).find((x) => x.startsWith('عربية')) ?? '';
const defs: string[] = [], links: string[] = [];
let skip = false;
for (const line of wt.split('\n')) {
const h = line.match(/^=+\s*(.*?)\s*=+\s*$/);
if (h) { skip = /ریشه|ترجم|برگردان|تصريف|تصریف|منابع|مشتق|نفس الجذر/.test(h[1]); continue; }
if (skip) continue;
const only = line.match(/^[#*]\s*'*\[\[([^\]|]+)\]\]'*\s*(?:\([^)]*\))?\.?\s*$/);
if (only) { links.push(only[1]); continue; }
const m = line.match(/^#(?![:*])\s*(.+)/);
if (!m) continue;
const t = unlink(stripTemplates(m[1])).replace(/^[\s.،:-]+|\s+$/g, '');
if (t.length >= 3 && !/^-+$/.test(t)) defs.push(t);
}
return { defs: defs.slice(0, 6), origin: null as string | null, links: links.slice(0, 5) };
}
export const ownEntry = (code: string, wt: string) => (code === 'ur' ? urduEntry(wt) : definitions(wt, code));
// pages of a MediaWiki XML dump (articles only, no redirects)
export function* dumpPages(xml: string) {
for (const m of xml.matchAll(/([\s\S]*?)<\/page>/g)) {
const p = m[1];
if (!/0<\/ns>/.test(p) || /([^<]*)<\/title>/)?.[1];
const text = p.match(/]*>([\s\S]*?)<\/text>/)?.[1];
if (title && text) yield { title: xmlText(title), text: xmlText(text) };
}
}
const xmlText = (s: string) => s.replace(/</g, '<').replace(/>/g, '>').replace(/"/g, '"').replace(/'/g, "'").replace(/&/g, '&');
// ---- import and sync ----
const meta = async (k: string) => (await pool.query('SELECT value FROM dict_meta WHERE name = $1', [k])).rows[0]?.value ?? null;
const setMeta = (k: string, v: string) =>
pool.query('INSERT INTO dict_meta (name, value) VALUES ($1, $2) ON CONFLICT (name) DO UPDATE SET value = $2', [k, v]);
async function insertRows(client: any, rows: unknown[][]) {
for (let i = 0; i < rows.length; i += 500) {
const chunk = rows.slice(i, i + 500), values: unknown[] = [];
const sql = chunk.map((r, j) => { values.push(...r); const o = j * 5; return `($${o + 1},$${o + 2},$${o + 3},$${o + 4},$${o + 5})`; });
await client.query(`INSERT INTO wiktionary (lang, source, title, key, data) VALUES ${sql.join(',')}`, values);
}
}
// replace one source's rows in a transaction (readers see the old data until it commits)
async function replace(lang: string, source: string, fill: (add: (title: string, data: object) => Promise, client: any) => Promise) {
const client = await pool.connect();
let n = 0;
try {
await client.query('BEGIN');
await client.query('DELETE FROM wiktionary WHERE lang = $1 AND source = $2', [lang, source]);
if (lang === 'ur') await client.query('DELETE FROM ur_glosses WHERE source = $1', [source]);
let batch: unknown[][] = [];
await fill(async (title, data) => {
batch.push([lang, source, title, key(title), data]); n++;
if (batch.length >= 2000) { await insertRows(client, batch); batch = []; }
}, client);
await insertRows(client, batch);
await client.query('COMMIT');
} catch (e) {
await client.query('ROLLBACK');
throw e;
} finally {
client.release();
}
return n;
}
async function lastModified(url: string) {
const res = await fetch(url, { method: 'HEAD', headers: { 'user-agent': UA } });
if (!res.ok) throw new Error(`${res.status} ${url}`);
return res.headers.get('last-modified') ?? '';
}
async function importKaikki(lang: string) {
const res = await fetch(KAIKKI(lang), { headers: { 'user-agent': UA } });
if (!res.ok || !res.body) throw new Error(`${res.status} ${KAIKKI(lang)}`);
return replace(lang, 'en', async (add, client) => {
const glossRows: string[][] = [];
for await (const line of createInterface({ input: Readable.fromWeb(res.body as any), crlfDelay: Infinity })) {
if (!line) continue;
const d = JSON.parse(line), e = kaikkiEntry(d);
if (!e.glosses.length && !e.ipa.length) continue;
await add(d.word, e);
if (lang === 'ur' && !e.formOf) for (const g of glossKeys(e.glosses.slice(0, 3))) glossRows.push([g, d.word]);
}
await insertGlosses(client, 'en', glossRows);
});
}
async function insertGlosses(client: any, source: string, rows: string[][]) {
for (let i = 0; i < rows.length; i += 1000) {
const chunk = rows.slice(i, i + 1000);
await client.query(`INSERT INTO ur_glosses (gloss, word, source) VALUES ${chunk.map((_, j) => `($${j * 2 + 1},$${j * 2 + 2},'${source}')`).join(',')}`, chunk.flat());
}
}
async function importDump(lang: string) {
const bz = spawn('sh', ['-c', `curl -sfL -A '${UA}' '${DUMP(lang)}' | bzcat`]);
return replace(lang, 'own', async (add, client) => {
let buf = '';
const glossRows: string[][] = []; // ur.wiktionary English entries -> the pivot table
for await (const chunk of bz.stdout.setEncoding('utf8')) {
buf += chunk;
const end = buf.lastIndexOf('');
if (end < 0) continue;
for (const p of dumpPages(buf.slice(0, end + 7))) {
if (lang === 'ur' && isEnglish(p.title)) {
for (const w of englishToUrdu(p.text)) glossRows.push([p.title.toLowerCase(), w]);
continue;
}
const e = ownEntry(lang, p.text);
if (e.defs.length || e.links.length || 'english' in e) await add(p.title, e);
}
buf = buf.slice(end + 7);
}
const code: number = await new Promise((r) => (bz.exitCode !== null ? r(bz.exitCode) : bz.on('close', r)));
if (code !== 0) throw new Error(`dump download/decompress failed for ${lang} (exit ${code})`);
await insertGlosses(client, 'own', glossRows);
});
}
// changes on ur./fa./ar.wiktionary since the last run, re-read from the API (titles in batches of 50)
async function recentChanges(lang: string) {
const api = `https://${lang}.wiktionary.org/w/api.php`;
const since = await meta(`rc:${lang}`), now = new Date().toISOString();
if (!since) return setMeta(`rc:${lang}`, now).then(() => 0); // first run: the dump is the baseline
const titles = new Set();
let cont: Record = {};
do {
const q = new URLSearchParams({ action: 'query', list: 'recentchanges', rcnamespace: '0', rctype: 'edit|new', rcprop: 'title',
rclimit: '500', rcdir: 'newer', rcstart: since, rcend: now, format: 'json', formatversion: '2', ...cont });
const d: any = await (await fetch(`${api}?${q}`, { headers: { 'user-agent': UA } })).json();
for (const c of d.query?.recentchanges ?? []) titles.add(c.title);
cont = d.continue ?? {};
} while (cont.rccontinue);
const list = [...titles];
for (let i = 0; i < list.length; i += 50) {
const q = new URLSearchParams({ action: 'query', prop: 'revisions', rvprop: 'content', rvslots: 'main',
titles: list.slice(i, i + 50).join('|'), format: 'json', formatversion: '2' });
const d: any = await (await fetch(api, { method: 'POST', body: q, headers: { 'user-agent': UA } })).json();
for (const p of d.query?.pages ?? []) {
const wt = p.revisions?.[0]?.slots?.main?.content;
if (lang === 'ur' && isEnglish(p.title)) {
const g = p.title.toLowerCase();
await pool.query(`DELETE FROM ur_glosses WHERE source = 'own' AND gloss = $1`, [g]);
for (const w of wt ? englishToUrdu(wt) : []) await pool.query(`INSERT INTO ur_glosses (gloss, word, source) VALUES ($1, $2, 'own')`, [g, w]);
continue;
}
await pool.query(`DELETE FROM wiktionary WHERE lang = $1 AND source = 'own' AND title = $2`, [lang, p.title]);
const e = wt && !/^#(REDIRECT|تحويل|تغییر)/i.test(wt) ? ownEntry(lang, wt) : null;
if (e && (e.defs.length || e.links.length || 'english' in e))
await pool.query(`INSERT INTO wiktionary (lang, source, title, key, data) VALUES ($1, 'own', $2, $3, $4)`, [lang, p.title, key(p.title), e]);
}
}
await setMeta(`rc:${lang}`, now);
return list.length;
}
// full re-import of each source whose upstream file changed, then the daily recent changes
export async function sync(log = console.log) {
for (const lang of LANGS) {
for (const [source, url, run] of [['en', KAIKKI(lang), importKaikki], ['own', DUMP(lang), importDump]] as const) {
const lm = await lastModified(url), k = `file:${lang}:${source}`;
if (lm && lm === (await meta(k))) continue;
const t = Date.now(), n = await run(lang);
await setMeta(k, lm);
if (source === 'own') await setMeta(`rc:${lang}`, new Date(lm).toISOString()); // changes after the dump
log(`${lang} ${source}: ${n} entries (${Math.round((Date.now() - t) / 1000)} s)`);
}
log(`${lang} recent changes: ${await recentChanges(lang)} pages`);
}
}
// ---- lookup ----
const page = (lang: string, source: string, title: string) =>
source === 'en' ? `https://en.wiktionary.org/wiki/${encodeURIComponent(title)}#${NAME[lang as 'ur']}` : `https://${lang}.wiktionary.org/wiki/${encodeURIComponent(title)}`;
export async function lookup(word: string) {
const k = key(word);
const { rows } = await pool.query('SELECT lang, source, title, data FROM wiktionary WHERE key = $1 ORDER BY id', [k]);
// ar.wiktionary pointer pages: follow to the diacritised entries
const pointers = rows.filter((r) => r.source === 'own' && !r.data.defs.length).flatMap((r) => r.data.links.map((l: string) => [r.lang, l]));
if (pointers.length) {
const more = await pool.query(
`SELECT lang, source, title, data FROM wiktionary WHERE source = 'own' AND (lang, title) IN (SELECT * FROM unnest($1::text[], $2::text[]))`,
[pointers.map((p) => p[0]), pointers.map((p) => p[1])]);
rows.push(...more.rows);
}
const langs = LANGS.map((code) => {
const en = rows.filter((r) => r.lang === code && r.source === 'en');
const own = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.defs.length);
const lemmas = en.filter((r) => !r.data.formOf), use = lemmas.length ? lemmas : en;
// readings: one spelling, different (unwritten) short vowels, e.g. ملک = مُلْک mulk, مَلِک malik, مِلْک milk
const readings: any[] = [];
for (const r of use) {
const id = r.data.tr ?? r.data.form ?? r.title;
let g = readings.find((x) => x.id === id);
if (!g) readings.push((g = { id, form: r.data.form ?? r.title, tr: r.data.tr, ipa: r.data.ipa, audio: r.data.audio, senses: [] }));
if (!g.ipa.length) g.ipa = r.data.ipa;
g.audio ??= r.data.audio;
if (r.data.glosses.length && g.senses.length < 3) g.senses.push({ pos: r.data.pos, defs: r.data.glosses.slice(0, 4) });
}
return {
code, readings: readings.slice(0, 5).map(({ id, ...x }) => x),
ety: use.map((r) => r.data.ety).find(Boolean) ?? null,
synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))].slice(0, 8),
meanings: own.flatMap((r) => r.data.defs).slice(0, 6), origin: own.find((r) => r.data.origin)?.data.origin ?? null,
source: own[0] ? page(code, 'own', own[0].title) : null,
// ur.wiktionary's English translation lines ("انگریزی : …")
translation: rows.filter((r) => r.lang === code && r.source === 'own' && r.data.english).map((r) => r.data.english),
en: en[0] ? page(code, 'en', en[0].title) : null,
};
}).filter((l) => l.meanings.length || l.readings.length || l.translation.length);
// Urdu translations through English (the primary English meaning's Urdu words): for the Urdu entry when it
// has no Urdu-language meaning, and for the Persian and Arabic entries when the word is not in Urdu at all
const inUrdu = langs.some((l) => l.code === 'ur');
await Promise.all(langs.map(async (l: any) => {
l.urdu = (l.code === 'ur' ? !l.meanings.length : !inUrdu) ? await viaEnglish(l.readings[0]?.senses[0]?.defs.slice(0, 2) ?? [], k) : [];
}));
const found = langs.length > 0;
// always: the same consonants with other long vowels (دل: دال، دول، دیل); not found: also stems and near spellings
const [variants, near] = await Promise.all([vowelVariants(k), found ? [] : similar(k)]);
return { word, langs, found, variants, similar: near.filter((n) => !variants.some((v) => v.title === n.title)) };
}
async function viaEnglish(glosses: string[], k: string) {
const g = glossKeys(glosses).slice(0, 4);
if (!g.length) return [];
const { rows } = await pool.query('SELECT gloss, word FROM ur_glosses WHERE gloss = ANY($1)', [g]);
const seen = new Set([k]);
return g.map((gloss) => ({
gloss, urdu: rows.filter((r) => r.gloss === gloss && !seen.has(key(r.word)) && seen.add(key(r.word))).map((r) => r.word).slice(0, 8),
})).filter((e) => e.urdu.length);
}
// ---- not found: similar words ----
// Urdu inflection endings (in key form) and the base forms to try: آنکھوں -> آنکھ, دیوانے -> دیوانه, جاتے -> جانا
const ENDINGS = ['یاں', 'ئیں', 'ؤں', 'وں', 'یں', 'گی', 'گا', 'گے', 'تا', 'تے', 'تی', 'نا', 'نے', 'نی', 'ے', 'ی', 'ا', 'و', 'ه'].map(key); // in key form (ں -> ن)
export function stems(k: string) {
const out = new Set();
for (const e of ENDINGS) {
if (!k.endsWith(e) || k.length - e.length < 2) continue;
const base = k.slice(0, -e.length);
for (const s of [base, base + 'ه', base + 'ا', base + 'نا']) if (s !== k) out.add(s);
}
return [...out];
}
// consonant skeleton as a regex: long vowels, heh and hamza optional (spellings add or drop them)
const SOFT = 'اوی\u0647ےءئ';
const esc = (c: string) => c.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
export function skeleton(k: string) {
const cs = [...k].filter((c) => !SOFT.includes(c));
if (cs.length < 2) return null;
const v = `[${SOFT}]*`;
return `^${v}${cs.map(esc).join(v)}${v}$`;
}
// shown titles without punctuation, dashes, spaces, tatweel or zero-width joiners
export const clean = (t: string) => t.replace(PUNCT, '').replace(/[\u0640\u200C\u200D]/g, '');
async function vowelVariants(k: string) {
const sk = skeleton(k);
if (!sk) return [];
const { rows } = await pool.query(
`SELECT key, (array_agg(title ORDER BY lang <> 'ur', lang <> 'fa', title))[1] AS title, array_agg(DISTINCT lang) AS langs
FROM wiktionary WHERE key ~ $1 AND key <> $2 GROUP BY key ORDER BY bool_or(lang = 'ur') DESC, similarity(key, $2) DESC LIMIT 10`, [sk, k]);
return rows.map((r) => ({ title: clean(r.title), langs: LANGS.filter((l) => r.langs.includes(l)) }));
}
async function similar(k: string) {
const st = stems(k), sk = skeleton(k), pre = [k, ...st].filter((s) => s.length >= 3).sort((a, b) => b.length - a.length)[0];
// one row per key: the Urdu spelling when there is one; Urdu entries first within each tier
const pick = `SELECT key, (array_agg(title ORDER BY lang <> 'ur', lang <> 'fa', title))[1] AS title, array_agg(DISTINCT lang) AS langs FROM wiktionary`;
const ur = `bool_or(lang = 'ur') DESC`;
const [a, b, c, d] = await Promise.all([
st.length ? pool.query(`${pick} WHERE key = ANY($1) GROUP BY key ORDER BY ${ur}`, [st]) : { rows: [] },
sk ? pool.query(`${pick} WHERE key ~ $1 AND key <> $2 GROUP BY key ORDER BY ${ur}, similarity(key, $2) DESC LIMIT 8`, [sk, k]) : { rows: [] },
pre ? pool.query(`${pick} WHERE key ~ $1 AND key <> $2 GROUP BY key ORDER BY ${ur}, length(key) LIMIT 8`, ['^' + [...pre].map(esc).join(''), k]) : { rows: [] },
pool.query(`${pick} WHERE key % $1 AND key <> $1 GROUP BY key ORDER BY ${ur}, similarity(key, $1) DESC LIMIT 8`, [k]),
]);
const seen = new Set(), out: { title: string; langs: string[] }[] = [];
for (const r of [...a.rows, ...b.rows, ...c.rows, ...d.rows])
if (!seen.has(r.key) && seen.add(r.key)) out.push({ title: clean(r.title), langs: LANGS.filter((l) => r.langs.includes(l)) });
return out.slice(0, 12);
}