- readings: one spelling, different short vowels (ملک: مَلَک، مَلِک، مِلک، مُلک), each with form, transliteration, IPA, audio and meanings - always: same consonants with other long vowels (consonant-skeleton regex) - not found: inflection stems (آنکھوں -> آنکھ, جاتے -> جانا), prefix regex, trigram similarity - Persian/Arabic words with no Urdu entry: Urdu translations via English first, then English, then their own language; filler glosses (especially, usually…) dropped - punctuation, dashes and spaces dropped from selected words and lookup keys; clean titles Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
384 lines
22 KiB
TypeScript
384 lines
22 KiB
TypeScript
// Word dictionary for the reading sidebar: the full Wiktionary data for Urdu, Persian and Arabic, in PostgreSQL.
|
||
// source 'en': en.wiktionary entries (English meanings, IPA, transliteration, audio, etymology, synonyms),
|
||
// from the kaikki.org Wiktextract extracts (weekly)
|
||
// source 'own': each language's own Wiktionary (ur., fa., ar.wiktionary): meanings in that language,
|
||
// from the Wikimedia dumps (twice a month) plus recent changes (daily)
|
||
// Urdu equivalents for words with no Urdu meaning are pivoted through English: ur_glosses maps English words to
|
||
// Urdu words, from the English glosses of en.wiktionary's Urdu entries and ur.wiktionary's English entries. Data: CC BY-SA, Wiktionary contributors.
|
||
import { spawn } from 'node:child_process';
|
||
import { createInterface } from 'node:readline';
|
||
import { Readable } from 'node:stream';
|
||
import { pool } from './db.ts';
|
||
|
||
export const LANGS = ['ur', 'fa', 'ar'] as const;
|
||
const NAME = { ur: 'Urdu', fa: 'Persian', ar: 'Arabic' } as const;
|
||
const UA = 'Divan/2 (https://github.com/anas-rashid/divan)';
|
||
const KAIKKI = (l: string) => `https://kaikki.org/dictionary/${NAME[l as 'ur']}/kaikki.org-dictionary-${NAME[l as 'ur']}.jsonl`;
|
||
const DUMP = (l: string) => `https://dumps.wikimedia.org/${l}wiktionary/latest/${l}wiktionary-latest-pages-articles.xml.bz2`;
|
||
const MARKS = /[ؐ-ًؚ-ٰٟۖ-ۭـ]/g;
|
||
|
||
// punctuation, dashes and spaces: dropped from words (بے ثبوت = بےثبوت, دل، = دل)
|
||
export const PUNCT = /[\s\-‐-―_.,،۔؛؟!?:;"'«»()[\]{}/\\*]+/g;
|
||
|
||
// spelling-insensitive key shared by Urdu, Persian and Arabic: no diacritics/tatweel/punctuation/spaces, one heh,
|
||
// one yeh, one kaf, plain alef, noon ghunna as noon (ناداں = نادان, قسمت = قِسْمَت, نگاہ = نگاه, معنی = معنى).
|
||
// ے and ھ stay distinct.
|
||
export const key = (w: string) =>
|
||
w.normalize('NFC').replace(MARKS, '').replace(PUNCT, '')
|
||
.replace(/[ہۂۃةه]/g, 'ه').replace(/[يىی]/g, 'ی')
|
||
.replace(/ك/g, 'ک').replace(/[أإٱ]/g, 'ا').replace(/ں/g, 'ن');
|
||
|
||
const unlink = (wt: string) => wt.replace(/\[\[(?:[^\]|]*\|)?([^\]]*)\]\]/g, '$1').replace(/'''?/g, '').trim();
|
||
const upload = (u?: string) => (u && u.startsWith('https://upload.wikimedia.org/') ? u : null);
|
||
|
||
// ---- parsing ----
|
||
|
||
// a kaikki.org (Wiktextract) line -> the fields the sidebar shows
|
||
export function kaikkiEntry(d: any) {
|
||
const senses = (d.senses ?? []).filter((s: any) => s.glosses?.length);
|
||
const sounds = d.sounds ?? [];
|
||
return {
|
||
pos: d.pos as string,
|
||
glosses: senses.map((s: any) => s.glosses.join('; ')).slice(0, 6) as string[],
|
||
formOf: senses.length > 0 && senses.every((s: any) => s.form_of || s.tags?.includes('form-of')),
|
||
ipa: [...new Set(sounds.map((s: any) => s.ipa).filter(Boolean))].slice(0, 2) as string[],
|
||
audio: sounds.map((s: any) => upload(s.mp3_url) ?? upload(s.ogg_url)).find(Boolean) ?? null,
|
||
tr: d.forms?.find((f: any) => f.tags?.includes('romanization'))?.form ?? null,
|
||
form: d.forms?.find((f: any) => f.tags?.includes('canonical'))?.form ?? null, // with short vowels: مُلْک
|
||
ety: d.etymology_text ? etymology(String(d.etymology_text)) : null,
|
||
synonyms: (d.synonyms ?? []).map((s: any) => s.word).filter(Boolean).slice(0, 8) as string[],
|
||
};
|
||
}
|
||
|
||
// etymology text without Wiktionary's "Etymology tree …" diagram summary; cut at 300 characters, never
|
||
// leaving half a surrogate pair
|
||
const etymology = (t: string) =>
|
||
(t.startsWith('Etymology tree') ? t.slice(Math.max(0, t.search(/\b(Borrowed|Inherited|From|Learned|Derived|Semi-learned|Calque|Compound)\b/))) : t)
|
||
.slice(0, 300).toWellFormed();
|
||
|
||
// English glosses of an Urdu entry as pivot keys: "fate, destiny" -> fate, destiny (short, lowercase)
|
||
export const glossKeys = (glosses: string[]) => [...new Set(glosses.flatMap((g) => g.replace(/\([^)]*\)/g, '').split(/[,;]/))
|
||
.map((s) => s.trim().toLowerCase().replace(/^(to|a|an|the) /, '')).filter((s) => /^[a-z][a-z' -]{1,30}$/.test(s) && s.split(' ').length <= 3 && !FILLER.has(s)))];
|
||
const FILLER = new Set(['especially', 'usually', 'often', 'also', 'etc', 'something', 'someone', 'chiefly', 'mainly', 'figuratively', 'literally', 'rare', 'archaic', 'obsolete', 'dated', 'poetic', 'colloquial', 'informal', 'formal', 'slang', 'by extension']);
|
||
|
||
// ur.wiktionary Urdu entries: numbered lines under ==معانی==, else "# " lines, else the opening prose;
|
||
// origin from "(عربی)" on the first line
|
||
export function urduEntry(wt: string) {
|
||
const m = wt.split(/==\s*معانی\s*==/)[1]?.split(/\n==/)[0] ?? '';
|
||
let defs = m.split('\n').map((l) => l.match(/^\s*(?:\d+[.)-]|#)\s*(.+)/)?.[1]).filter(Boolean).map((l) => unlink(l!));
|
||
if (!defs.length) defs = definitions(wt, 'ur').defs;
|
||
if (!defs.length) {
|
||
const prose = wt.split('\n').find((l) => l.trim().startsWith("'''") && l.length > 20);
|
||
if (prose) defs = [unlink(stripTemplates(prose)).slice(0, 300)];
|
||
}
|
||
const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null;
|
||
const english = wt.match(/^\s*انگریزی\s*:\s*([A-Za-z].+)$/m)?.[1].trim() ?? null; // "== تراجم ==" pages
|
||
return { defs: defs.slice(0, 6), origin, links: [] as string[], ...(english && { english }) };
|
||
}
|
||
|
||
// ur.wiktionary English entries ("north": "# [[شمالی]]۔"): the Urdu words, for the pivot through English
|
||
export const isEnglish = (title: string) => /^[A-Za-z][A-Za-z' -]*$/.test(title);
|
||
export function englishToUrdu(wt: string) {
|
||
const lines = wt.split('\n').filter((l) => /^#(?![:*])/.test(l)).join(' ');
|
||
return [...new Set([...lines.matchAll(/\[\[(?:[^\]|]*\|)?([^\]]+)\]\]/g)].map((m) => m[1].trim())
|
||
.filter((w) => /^[\p{Script=Arabic}\s]+$/u.test(w)))].slice(0, 10);
|
||
}
|
||
const stripTemplates = (t: string) => {
|
||
while (/\{\{[^{}]*\}\}/.test(t)) t = t.replace(/\{\{[^{}]*\}\}/g, '');
|
||
return t;
|
||
};
|
||
|
||
// fa./ar.wiktionary: "# definition" lines outside etymology/translation sections; on ar.wiktionary only the
|
||
// Arabic section, and undiacritised pages that just point to entries ("* [[عِشْق]]") give those links
|
||
export function definitions(wt: string, code: string) {
|
||
if (code === 'ar' && wt.includes('{{اللغة|')) wt = wt.split(/==\s*\{\{اللغة\|/).find((x) => x.startsWith('عربية')) ?? '';
|
||
const defs: string[] = [], links: string[] = [];
|
||
let skip = false;
|
||
for (const line of wt.split('\n')) {
|
||
const h = line.match(/^=+\s*(.*?)\s*=+\s*$/);
|
||
if (h) { skip = /ریشه|ترجم|برگردان|تصريف|تصریف|منابع|مشتق|نفس الجذر/.test(h[1]); continue; }
|
||
if (skip) continue;
|
||
const only = line.match(/^[#*]\s*'*\[\[([^\]|]+)\]\]'*\s*(?:\([^)]*\))?\.?\s*$/);
|
||
if (only) { links.push(only[1]); continue; }
|
||
const m = line.match(/^#(?![:*])\s*(.+)/);
|
||
if (!m) continue;
|
||
const t = unlink(stripTemplates(m[1])).replace(/^[\s.،:-]+|\s+$/g, '');
|
||
if (t.length >= 3 && !/^-+$/.test(t)) defs.push(t);
|
||
}
|
||
return { defs: defs.slice(0, 6), origin: null as string | null, links: links.slice(0, 5) };
|
||
}
|
||
|
||
export const ownEntry = (code: string, wt: string) => (code === 'ur' ? urduEntry(wt) : definitions(wt, code));
|
||
|
||
// pages of a MediaWiki XML dump (articles only, no redirects)
|
||
export function* dumpPages(xml: string) {
|
||
for (const m of xml.matchAll(/<page>([\s\S]*?)<\/page>/g)) {
|
||
const p = m[1];
|
||
if (!/<ns>0<\/ns>/.test(p) || /<redirect /.test(p)) continue;
|
||
const title = p.match(/<title>([^<]*)<\/title>/)?.[1];
|
||
const text = p.match(/<text[^>]*>([\s\S]*?)<\/text>/)?.[1];
|
||
if (title && text) yield { title: xmlText(title), text: xmlText(text) };
|
||
}
|
||
}
|
||
const xmlText = (s: string) => s.replace(/</g, '<').replace(/>/g, '>').replace(/"/g, '"').replace(/'/g, "'").replace(/&/g, '&');
|
||
|
||
// ---- import and sync ----
|
||
|
||
const meta = async (k: string) => (await pool.query('SELECT value FROM dict_meta WHERE name = $1', [k])).rows[0]?.value ?? null;
|
||
const setMeta = (k: string, v: string) =>
|
||
pool.query('INSERT INTO dict_meta (name, value) VALUES ($1, $2) ON CONFLICT (name) DO UPDATE SET value = $2', [k, v]);
|
||
|
||
async function insertRows(client: any, rows: unknown[][]) {
|
||
for (let i = 0; i < rows.length; i += 500) {
|
||
const chunk = rows.slice(i, i + 500), values: unknown[] = [];
|
||
const sql = chunk.map((r, j) => { values.push(...r); const o = j * 5; return `($${o + 1},$${o + 2},$${o + 3},$${o + 4},$${o + 5})`; });
|
||
await client.query(`INSERT INTO wiktionary (lang, source, title, key, data) VALUES ${sql.join(',')}`, values);
|
||
}
|
||
}
|
||
|
||
// replace one source's rows in a transaction (readers see the old data until it commits)
|
||
async function replace(lang: string, source: string, fill: (add: (title: string, data: object) => Promise<void>, client: any) => Promise<void>) {
|
||
const client = await pool.connect();
|
||
let n = 0;
|
||
try {
|
||
await client.query('BEGIN');
|
||
await client.query('DELETE FROM wiktionary WHERE lang = $1 AND source = $2', [lang, source]);
|
||
if (lang === 'ur') await client.query('DELETE FROM ur_glosses WHERE source = $1', [source]);
|
||
let batch: unknown[][] = [];
|
||
await fill(async (title, data) => {
|
||
batch.push([lang, source, title, key(title), data]); n++;
|
||
if (batch.length >= 2000) { await insertRows(client, batch); batch = []; }
|
||
}, client);
|
||
await insertRows(client, batch);
|
||
await client.query('COMMIT');
|
||
} catch (e) {
|
||
await client.query('ROLLBACK');
|
||
throw e;
|
||
} finally {
|
||
client.release();
|
||
}
|
||
return n;
|
||
}
|
||
|
||
async function lastModified(url: string) {
|
||
const res = await fetch(url, { method: 'HEAD', headers: { 'user-agent': UA } });
|
||
if (!res.ok) throw new Error(`${res.status} ${url}`);
|
||
return res.headers.get('last-modified') ?? '';
|
||
}
|
||
|
||
async function importKaikki(lang: string) {
|
||
const res = await fetch(KAIKKI(lang), { headers: { 'user-agent': UA } });
|
||
if (!res.ok || !res.body) throw new Error(`${res.status} ${KAIKKI(lang)}`);
|
||
return replace(lang, 'en', async (add, client) => {
|
||
const glossRows: string[][] = [];
|
||
for await (const line of createInterface({ input: Readable.fromWeb(res.body as any), crlfDelay: Infinity })) {
|
||
if (!line) continue;
|
||
const d = JSON.parse(line), e = kaikkiEntry(d);
|
||
if (!e.glosses.length && !e.ipa.length) continue;
|
||
await add(d.word, e);
|
||
if (lang === 'ur' && !e.formOf) for (const g of glossKeys(e.glosses.slice(0, 3))) glossRows.push([g, d.word]);
|
||
}
|
||
await insertGlosses(client, 'en', glossRows);
|
||
});
|
||
}
|
||
|
||
async function insertGlosses(client: any, source: string, rows: string[][]) {
|
||
for (let i = 0; i < rows.length; i += 1000) {
|
||
const chunk = rows.slice(i, i + 1000);
|
||
await client.query(`INSERT INTO ur_glosses (gloss, word, source) VALUES ${chunk.map((_, j) => `($${j * 2 + 1},$${j * 2 + 2},'${source}')`).join(',')}`, chunk.flat());
|
||
}
|
||
}
|
||
|
||
async function importDump(lang: string) {
|
||
const bz = spawn('sh', ['-c', `curl -sfL -A '${UA}' '${DUMP(lang)}' | bzcat`]);
|
||
return replace(lang, 'own', async (add, client) => {
|
||
let buf = '';
|
||
const glossRows: string[][] = []; // ur.wiktionary English entries -> the pivot table
|
||
for await (const chunk of bz.stdout.setEncoding('utf8')) {
|
||
buf += chunk;
|
||
const end = buf.lastIndexOf('</page>');
|
||
if (end < 0) continue;
|
||
for (const p of dumpPages(buf.slice(0, end + 7))) {
|
||
if (lang === 'ur' && isEnglish(p.title)) {
|
||
for (const w of englishToUrdu(p.text)) glossRows.push([p.title.toLowerCase(), w]);
|
||
continue;
|
||
}
|
||
const e = ownEntry(lang, p.text);
|
||
if (e.defs.length || e.links.length || 'english' in e) await add(p.title, e);
|
||
}
|
||
buf = buf.slice(end + 7);
|
||
}
|
||
const code: number = await new Promise((r) => (bz.exitCode !== null ? r(bz.exitCode) : bz.on('close', r)));
|
||
if (code !== 0) throw new Error(`dump download/decompress failed for ${lang} (exit ${code})`);
|
||
await insertGlosses(client, 'own', glossRows);
|
||
});
|
||
}
|
||
|
||
// changes on ur./fa./ar.wiktionary since the last run, re-read from the API (titles in batches of 50)
|
||
async function recentChanges(lang: string) {
|
||
const api = `https://${lang}.wiktionary.org/w/api.php`;
|
||
const since = await meta(`rc:${lang}`), now = new Date().toISOString();
|
||
if (!since) return setMeta(`rc:${lang}`, now).then(() => 0); // first run: the dump is the baseline
|
||
const titles = new Set<string>();
|
||
let cont: Record<string, string> = {};
|
||
do {
|
||
const q = new URLSearchParams({ action: 'query', list: 'recentchanges', rcnamespace: '0', rctype: 'edit|new', rcprop: 'title',
|
||
rclimit: '500', rcdir: 'newer', rcstart: since, rcend: now, format: 'json', formatversion: '2', ...cont });
|
||
const d: any = await (await fetch(`${api}?${q}`, { headers: { 'user-agent': UA } })).json();
|
||
for (const c of d.query?.recentchanges ?? []) titles.add(c.title);
|
||
cont = d.continue ?? {};
|
||
} while (cont.rccontinue);
|
||
const list = [...titles];
|
||
for (let i = 0; i < list.length; i += 50) {
|
||
const q = new URLSearchParams({ action: 'query', prop: 'revisions', rvprop: 'content', rvslots: 'main',
|
||
titles: list.slice(i, i + 50).join('|'), format: 'json', formatversion: '2' });
|
||
const d: any = await (await fetch(api, { method: 'POST', body: q, headers: { 'user-agent': UA } })).json();
|
||
for (const p of d.query?.pages ?? []) {
|
||
const wt = p.revisions?.[0]?.slots?.main?.content;
|
||
if (lang === 'ur' && isEnglish(p.title)) {
|
||
const g = p.title.toLowerCase();
|
||
await pool.query(`DELETE FROM ur_glosses WHERE source = 'own' AND gloss = $1`, [g]);
|
||
for (const w of wt ? englishToUrdu(wt) : []) await pool.query(`INSERT INTO ur_glosses (gloss, word, source) VALUES ($1, $2, 'own')`, [g, w]);
|
||
continue;
|
||
}
|
||
await pool.query(`DELETE FROM wiktionary WHERE lang = $1 AND source = 'own' AND title = $2`, [lang, p.title]);
|
||
const e = wt && !/^#(REDIRECT|تحويل|تغییر)/i.test(wt) ? ownEntry(lang, wt) : null;
|
||
if (e && (e.defs.length || e.links.length || 'english' in e))
|
||
await pool.query(`INSERT INTO wiktionary (lang, source, title, key, data) VALUES ($1, 'own', $2, $3, $4)`, [lang, p.title, key(p.title), e]);
|
||
}
|
||
}
|
||
await setMeta(`rc:${lang}`, now);
|
||
return list.length;
|
||
}
|
||
|
||
// full re-import of each source whose upstream file changed, then the daily recent changes
|
||
export async function sync(log = console.log) {
|
||
for (const lang of LANGS) {
|
||
for (const [source, url, run] of [['en', KAIKKI(lang), importKaikki], ['own', DUMP(lang), importDump]] as const) {
|
||
const lm = await lastModified(url), k = `file:${lang}:${source}`;
|
||
if (lm && lm === (await meta(k))) continue;
|
||
const t = Date.now(), n = await run(lang);
|
||
await setMeta(k, lm);
|
||
if (source === 'own') await setMeta(`rc:${lang}`, new Date(lm).toISOString()); // changes after the dump
|
||
log(`${lang} ${source}: ${n} entries (${Math.round((Date.now() - t) / 1000)} s)`);
|
||
}
|
||
log(`${lang} recent changes: ${await recentChanges(lang)} pages`);
|
||
}
|
||
}
|
||
|
||
// ---- lookup ----
|
||
|
||
const page = (lang: string, source: string, title: string) =>
|
||
source === 'en' ? `https://en.wiktionary.org/wiki/${encodeURIComponent(title)}#${NAME[lang as 'ur']}` : `https://${lang}.wiktionary.org/wiki/${encodeURIComponent(title)}`;
|
||
|
||
export async function lookup(word: string) {
|
||
const k = key(word);
|
||
const { rows } = await pool.query('SELECT lang, source, title, data FROM wiktionary WHERE key = $1 ORDER BY id', [k]);
|
||
// ar.wiktionary pointer pages: follow to the diacritised entries
|
||
const pointers = rows.filter((r) => r.source === 'own' && !r.data.defs.length).flatMap((r) => r.data.links.map((l: string) => [r.lang, l]));
|
||
if (pointers.length) {
|
||
const more = await pool.query(
|
||
`SELECT lang, source, title, data FROM wiktionary WHERE source = 'own' AND (lang, title) IN (SELECT * FROM unnest($1::text[], $2::text[]))`,
|
||
[pointers.map((p) => p[0]), pointers.map((p) => p[1])]);
|
||
rows.push(...more.rows);
|
||
}
|
||
const langs = LANGS.map((code) => {
|
||
const en = rows.filter((r) => r.lang === code && r.source === 'en');
|
||
const own = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.defs.length);
|
||
const lemmas = en.filter((r) => !r.data.formOf), use = lemmas.length ? lemmas : en;
|
||
// readings: one spelling, different (unwritten) short vowels, e.g. ملک = مُلْک mulk, مَلِک malik, مِلْک milk
|
||
const readings: any[] = [];
|
||
for (const r of use) {
|
||
const id = r.data.tr ?? r.data.form ?? r.title;
|
||
let g = readings.find((x) => x.id === id);
|
||
if (!g) readings.push((g = { id, form: r.data.form ?? r.title, tr: r.data.tr, ipa: r.data.ipa, audio: r.data.audio, senses: [] }));
|
||
if (!g.ipa.length) g.ipa = r.data.ipa;
|
||
g.audio ??= r.data.audio;
|
||
if (r.data.glosses.length && g.senses.length < 3) g.senses.push({ pos: r.data.pos, defs: r.data.glosses.slice(0, 4) });
|
||
}
|
||
return {
|
||
code, readings: readings.slice(0, 5).map(({ id, ...x }) => x),
|
||
ety: use.map((r) => r.data.ety).find(Boolean) ?? null,
|
||
synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))].slice(0, 8),
|
||
meanings: own.flatMap((r) => r.data.defs).slice(0, 6), origin: own.find((r) => r.data.origin)?.data.origin ?? null,
|
||
source: own[0] ? page(code, 'own', own[0].title) : null,
|
||
// ur.wiktionary's English translation lines ("انگریزی : …")
|
||
translation: rows.filter((r) => r.lang === code && r.source === 'own' && r.data.english).map((r) => r.data.english),
|
||
en: en[0] ? page(code, 'en', en[0].title) : null,
|
||
};
|
||
}).filter((l) => l.meanings.length || l.readings.length || l.translation.length);
|
||
// Urdu translations through English (the primary English meaning's Urdu words): for the Urdu entry when it
|
||
// has no Urdu-language meaning, and for the Persian and Arabic entries when the word is not in Urdu at all
|
||
const inUrdu = langs.some((l) => l.code === 'ur');
|
||
await Promise.all(langs.map(async (l: any) => {
|
||
l.urdu = (l.code === 'ur' ? !l.meanings.length : !inUrdu) ? await viaEnglish(l.readings[0]?.senses[0]?.defs.slice(0, 2) ?? [], k) : [];
|
||
}));
|
||
const found = langs.length > 0;
|
||
// always: the same consonants with other long vowels (دل: دال، دول، دیل); not found: also stems and near spellings
|
||
const [variants, near] = await Promise.all([vowelVariants(k), found ? [] : similar(k)]);
|
||
return { word, langs, found, variants, similar: near.filter((n) => !variants.some((v) => v.title === n.title)) };
|
||
}
|
||
|
||
async function viaEnglish(glosses: string[], k: string) {
|
||
const g = glossKeys(glosses).slice(0, 4);
|
||
if (!g.length) return [];
|
||
const { rows } = await pool.query('SELECT gloss, word FROM ur_glosses WHERE gloss = ANY($1)', [g]);
|
||
const seen = new Set([k]);
|
||
return g.map((gloss) => ({
|
||
gloss, urdu: rows.filter((r) => r.gloss === gloss && !seen.has(key(r.word)) && seen.add(key(r.word))).map((r) => r.word).slice(0, 8),
|
||
})).filter((e) => e.urdu.length);
|
||
}
|
||
|
||
// ---- not found: similar words ----
|
||
|
||
// Urdu inflection endings (in key form) and the base forms to try: آنکھوں -> آنکھ, دیوانے -> دیوانه, جاتے -> جانا
|
||
const ENDINGS = ['یاں', 'ئیں', 'ؤں', 'وں', 'یں', 'گی', 'گا', 'گے', 'تا', 'تے', 'تی', 'نا', 'نے', 'نی', 'ے', 'ی', 'ا', 'و', 'ه'].map(key); // in key form (ں -> ن)
|
||
export function stems(k: string) {
|
||
const out = new Set<string>();
|
||
for (const e of ENDINGS) {
|
||
if (!k.endsWith(e) || k.length - e.length < 2) continue;
|
||
const base = k.slice(0, -e.length);
|
||
for (const s of [base, base + 'ه', base + 'ا', base + 'نا']) if (s !== k) out.add(s);
|
||
}
|
||
return [...out];
|
||
}
|
||
|
||
// consonant skeleton as a regex: long vowels, heh and hamza optional (spellings add or drop them)
|
||
const SOFT = 'اوی\u0647ےءئ';
|
||
const esc = (c: string) => c.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
||
export function skeleton(k: string) {
|
||
const cs = [...k].filter((c) => !SOFT.includes(c));
|
||
if (cs.length < 2) return null;
|
||
const v = `[${SOFT}]*`;
|
||
return `^${v}${cs.map(esc).join(v)}${v}$`;
|
||
}
|
||
|
||
// shown titles without punctuation, dashes, spaces, tatweel or zero-width joiners
|
||
export const clean = (t: string) => t.replace(PUNCT, '').replace(/[\u0640\u200C\u200D]/g, '');
|
||
|
||
async function vowelVariants(k: string) {
|
||
const sk = skeleton(k);
|
||
if (!sk) return [];
|
||
const { rows } = await pool.query(
|
||
`SELECT key, (array_agg(title ORDER BY lang <> 'ur', lang <> 'fa', title))[1] AS title, array_agg(DISTINCT lang) AS langs
|
||
FROM wiktionary WHERE key ~ $1 AND key <> $2 GROUP BY key ORDER BY bool_or(lang = 'ur') DESC, similarity(key, $2) DESC LIMIT 10`, [sk, k]);
|
||
return rows.map((r) => ({ title: clean(r.title), langs: LANGS.filter((l) => r.langs.includes(l)) }));
|
||
}
|
||
|
||
async function similar(k: string) {
|
||
const st = stems(k), sk = skeleton(k), pre = [k, ...st].filter((s) => s.length >= 3).sort((a, b) => b.length - a.length)[0];
|
||
// one row per key: the Urdu spelling when there is one; Urdu entries first within each tier
|
||
const pick = `SELECT key, (array_agg(title ORDER BY lang <> 'ur', lang <> 'fa', title))[1] AS title, array_agg(DISTINCT lang) AS langs FROM wiktionary`;
|
||
const ur = `bool_or(lang = 'ur') DESC`;
|
||
const [a, b, c, d] = await Promise.all([
|
||
st.length ? pool.query(`${pick} WHERE key = ANY($1) GROUP BY key ORDER BY ${ur}`, [st]) : { rows: [] },
|
||
sk ? pool.query(`${pick} WHERE key ~ $1 AND key <> $2 GROUP BY key ORDER BY ${ur}, similarity(key, $2) DESC LIMIT 8`, [sk, k]) : { rows: [] },
|
||
pre ? pool.query(`${pick} WHERE key ~ $1 AND key <> $2 GROUP BY key ORDER BY ${ur}, length(key) LIMIT 8`, ['^' + [...pre].map(esc).join(''), k]) : { rows: [] },
|
||
pool.query(`${pick} WHERE key % $1 AND key <> $1 GROUP BY key ORDER BY ${ur}, similarity(key, $1) DESC LIMIT 8`, [k]),
|
||
]);
|
||
const seen = new Set<string>(), out: { title: string; langs: string[] }[] = [];
|
||
for (const r of [...a.rows, ...b.rows, ...c.rows, ...d.rows])
|
||
if (!seen.has(r.key) && seen.add(r.key)) out.push({ title: clean(r.title), langs: LANGS.filter((l) => r.langs.includes(l)) });
|
||
return out.slice(0, 12);
|
||
}
|