Word dictionary: full Wiktionary data for Urdu, Persian and Arabic in PostgreSQL, kept in sync

- en.wiktionary entries via kaikki.org (meanings, IPA, transliteration, audio, etymology, synonyms)
- ur./fa./ar.wiktionary dumps for meanings in each language (incl. more Urdu entry formats)
- English->Urdu pivot table from Urdu entries' glosses and ur.wiktionary's English entries
- dict-sync.ts: re-imports sources when upstream publishes new files, applies daily recent changes;
  run from deploy/sync.sh
- lookups now read the database (2-7 ms) instead of calling Wikimedia live

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Anas Rashid 2026-10-08 21:48:56 +02:00
parent ef618094d5
commit 7fcc0ee7fa
9 changed files with 381 additions and 152 deletions

View File

@ -42,6 +42,8 @@ The import upserts, so re-running it after a divan-data sync applies the changes
15 3 * * * DATABASE_URL=postgres://divan:...@localhost:5432/divan /opt/divan/deploy/sync.sh >> /var/log/divan-sync.log 2>&1
```
The same script keeps the word dictionary current (`npm run dict-sync` in `api/`): the full Wiktionary data for Urdu, Persian and Arabic (English Wiktionary via [kaikki.org](https://kaikki.org), and the Urdu, Persian and Arabic Wiktionary dumps) is re-imported when upstream publishes new files, and each day's Wiktionary edits are applied from recent changes. The first run downloads about 700 MB.
Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DIVAN_APP_DIR` (default: this repo), `DIVAN_DATA_PUSH=1` to also commit and push data changes (needs git push access). Needs git, python3 and Node 24+.
## API
@ -51,7 +53,7 @@ Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DI
| `GET /api/poets` | all poets |
| `GET /api/page?url=/p238/...` | the poet, category or poem at a site URL (breadcrumbs, children, verses, prev/next) |
| `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised; exact phrase first; each with the best-matching couplet or paragraph (`snippet`) |
| `GET /api/word?w=` | one word's meanings and pronunciation from Wiktionary (Urdu, Persian, Arabic in that order; English meanings; Urdu equivalents via English when Urdu Wiktionary has none), cached 30 days |
| `GET /api/word?w=` | one word's meanings and pronunciation from the local Wiktionary data (Urdu, Persian, Arabic in that order; English meanings; Urdu equivalents via English when Urdu Wiktionary has none) |
| `GET /health` | database check |
Search normalises both stored text and queries: Arabic ي/ك/ه → Urdu ی/ک/ہ, ۂ/ۓ, diacritics and the Urdu full stop removed; do-chashmi ھ stays distinct.

View File

@ -9,7 +9,8 @@
"scripts": {
"start": "node src/server.ts",
"import": "node src/import.ts",
"test": "node --test src/*.test.ts"
"test": "node --test src/*.test.ts",
"dict-sync": "node src/dict-sync.ts"
},
"dependencies": {
"fastify": "^5.12.5",

10
api/src/dict-sync.ts Normal file
View File

@ -0,0 +1,10 @@
// Import / sync the Wiktionary data (see dictionary.ts). First run imports everything (about 1 GB of downloads);
// later runs re-import only sources that changed upstream and apply the day's Wiktionary edits.
// node src/dict-sync.ts
import { readFile } from 'node:fs/promises';
import { pool } from './db.ts';
import { sync } from './dictionary.ts';
await pool.query(await readFile(new URL('../../db/schema.sql', import.meta.url), 'utf8'));
await sync((line) => console.log(`${new Date().toISOString()} ${line}`));
await pool.end();

View File

@ -1,34 +1,67 @@
import { test } from 'node:test';
import assert from 'node:assert/strict';
import { forms, urduEntry, pronunciations } from './dictionary.ts';
import { key, kaikkiEntry, glossKeys, urduEntry, definitions, dumpPages } from './dictionary.ts';
test('spelling forms: diacritics dropped, final noon ghunna tried as noon', () => {
assert.deepEqual(forms('ناداں'), ['ناداں', 'نادان']);
assert.deepEqual(forms('وِصال'), ['وِصال', 'وصال']);
test('lookup key: spelling variants across Urdu, Persian and Arabic meet', () => {
const same = (a: string, b: string) => assert.equal(key(a), key(b), `${a} vs ${b}`);
same('ناداں', 'نادان');
same('قِسْمَت', 'قسمت');
same('نگاہ', 'نگاه');
same('معنی', 'معنى');
same('كتاب', 'کتاب');
assert.notEqual(key('کھ'), key('کہ'), 'do-chashmi heh stays distinct');
assert.notEqual(key('دیوانے'), key('دیوانی'), 'bari ye stays distinct');
});
test('kaikki entry: glosses, IPA, audio, romanisation, form-of', () => {
const e = kaikkiEntry({
word: 'قسمت', pos: 'noun', etymology_text: 'From Arabic',
senses: [{ glosses: ['fate, destiny'] }, { glosses: ['division'] }, { links: [] }],
sounds: [{ ipa: '/qɪs.mət̪/' }, { audio: 'x.wav', mp3_url: 'https://upload.wikimedia.org/x.mp3' }, { mp3_url: 'https://evil.example/x.mp3' }],
forms: [{ form: 'قِسْمَت', tags: ['canonical'] }, { form: 'qismat', tags: ['romanization'] }],
});
assert.deepEqual(e, { pos: 'noun', glosses: ['fate, destiny', 'division'], formOf: false, ipa: ['/qɪs.mət̪/'],
audio: 'https://upload.wikimedia.org/x.mp3', tr: 'qismat', ety: 'From Arabic', synonyms: [] });
assert.equal(kaikkiEntry({ word: 'x', pos: 'noun', senses: [{ glosses: ['plural of y'], tags: ['form-of'] }] }).formOf, true);
});
test('pivot keys from English glosses', () => {
assert.deepEqual(glossKeys(['fate, destiny', 'to love (someone)', 'a very long description of something that is not a key']), ['fate', 'destiny', 'love']);
});
test('ur.wiktionary entry: meanings and origin', () => {
const wt = "وِصال {وِصال} ([[عربی]])\n\nاسم [[نکرہ]]\n\n==معانی==\n\n1. بھیٹ، ملاقات، [[عاشق و معشوق]] کی محبت۔\n\n==مرکبات==\n\nوِصال کا دِن";
assert.deepEqual(urduEntry(wt), { meanings: ['بھیٹ، ملاقات، عاشق و معشوق کی محبت۔'], origin: 'عربی' });
const wt = 'وِصال {وِصال} ([[عربی]])\n\nاسم [[نکرہ]]\n\n==معانی==\n\n1. بھیٹ، ملاقات، [[عاشق و معشوق]] کی محبت۔\n\n==مرکبات==\n\nوِصال کا دِن';
assert.deepEqual(urduEntry(wt), { defs: ['بھیٹ، ملاقات، عاشق و معشوق کی محبت۔'], origin: 'عربی', links: [] });
});
test('en.wiktionary HTML: IPA, transliteration and audio per language', () => {
const html = '<h2 data-mw-wikitext="" id="Persian">Persian</h2><span class="IPA nowrap">/qis.ˈmat/</span><span class="IPA">-at</span>' +
'<span lang="fa-Latn" class="headword-tr manual-tr tr Latn" dir="ltr">qismat</span><audio><source src="//upload.wikimedia.org/a/b.ogg"></audio>' +
'<h2 id="Urdu">Urdu</h2><span class="IPA nowrap">/qɪs.mət̪/</span><span lang="ur-Latn" class="headword-tr tr Latn">qismat</span>';
assert.deepEqual(pronunciations(html), {
fa: { ipa: ['/qis.ˈmat/'], tr: 'qismat', audio: 'https://upload.wikimedia.org/a/b.ogg' },
ur: { ipa: ['/qɪs.mət̪/'], tr: 'qismat', audio: null },
});
});
test('fa./ar.wiktionary definitions: skip etymology, follow Arabic pointers', async () => {
const { definitions, formsIn } = await import('./dictionary.ts');
test('fa./ar.wiktionary definitions: skip etymology, Arabic pointers', () => {
const fa = '==فارسی==\n===ریشه‌شناسی===\n* [[عربی]]\n# قسمة\n===اسم===\n#بهره، نصیب.\n#:example\n#سرنوشت، تقدیر.\n====برگردان‌ها====\n# x y z';
assert.deepEqual(definitions(fa, 'fa').defs, ['بهره، نصیب.', 'سرنوشت، تقدیر.']);
const ar = '== {{اللغة|عربية}} ==\nهل تقصد:\n* [[عِشْق]]\n* [[عَشَقَ]]\n----\n== {{اللغة|أردية}} ==\n# [[x]]';
assert.deepEqual(definitions(ar, 'ar'), { defs: [], links: ['عِشْق', 'عَشَقَ'] });
assert.deepEqual(definitions(ar, 'ar'), { defs: [], origin: null, links: ['عِشْق', 'عَشَقَ'] });
assert.deepEqual(definitions('# {{مصدر|عَشِقَ}}.\n# فرط الحب.', 'ar').defs, ['فرط الحب.']);
assert.deepEqual(formsIn('fa', 'نگاہ'), ['نگاه']);
assert.deepEqual(formsIn('ar', 'معنی'), ['معني', 'معنى']);
});
test('dump pages: articles only, entities decoded, redirects skipped', () => {
const xml = '<page><title>عشق</title><ns>0</ns><revision><text bytes="9">a &amp; b &lt;x&gt;</text></revision></page>' +
'<page><title>Talk:x</title><ns>1</ns><text>t</text></page><page><title>r</title><ns>0</ns><redirect title="y" /><text>#R</text></page>';
assert.deepEqual([...dumpPages(xml)], [{ title: 'عشق', text: 'a & b <x>' }]);
});
test('etymology cut never leaves half a surrogate pair', () => {
const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'a'.repeat(299) + '𑀭𑀭' }).ety!;
assert.ok(ety.isWellFormed());
});
test('ur.wiktionary: English entries give Urdu words; Urdu entries fall back to # lines or prose', async () => {
const { englishToUrdu, isEnglish } = await import('./dictionary.ts');
assert.ok(isEnglish('north') && !isEnglish('شمال'));
assert.deepEqual(englishToUrdu('==انگریزی==\n===صفت===\n{{en-adjective}}\n# [[شمالی]]۔\n# [[شمال]] کی [[جانب]]۔\n#: [[x]]'), ['شمالی', 'شمال', 'جانب']);
assert.deepEqual(urduEntry('===اسم===\n# [[بہار]] کا [[موسم]]۔').defs, ['بہار کا موسم۔']);
assert.deepEqual(urduEntry("'''پھوڑی''' اس چادر کو کہتے ہیں جس پر بیٹھتے ہیں۔").defs, ['پھوڑی اس چادر کو کہتے ہیں جس پر بیٹھتے ہیں۔']);
});
test('etymology drops the "Etymology tree" summary', () => {
const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'Etymology tree Arabic قَسَمَ (qasama)bor. Urdu قِسْمَت Borrowed from Classical Persian قِسْمَت (qismat).' }).ety;
assert.equal(ety, 'Borrowed from Classical Persian قِسْمَت (qismat).');
});

View File

@ -1,48 +1,89 @@
// Word lookup for the reading sidebar, from Wiktionary (CC BY-SA). For Urdu, Persian and Arabic, in that order:
// - meanings in the language itself from its own Wiktionary (ur., fa., ar.wiktionary)
// - English meanings and pronunciation (IPA, transliteration, audio) from en.wiktionary
// With no Urdu meanings, Urdu equivalents are pivoted through English (the English glosses' translation tables).
// Found words are cached in PostgreSQL (dictionary table) for 30 days.
// Word dictionary for the reading sidebar: the full Wiktionary data for Urdu, Persian and Arabic, in PostgreSQL.
// source 'en': en.wiktionary entries (English meanings, IPA, transliteration, audio, etymology, synonyms),
// from the kaikki.org Wiktextract extracts (weekly)
// source 'own': each language's own Wiktionary (ur., fa., ar.wiktionary): meanings in that language,
// from the Wikimedia dumps (twice a month) plus recent changes (daily)
// Urdu equivalents for words with no Urdu meaning are pivoted through English: ur_glosses maps English words to
// Urdu words, from the English glosses of en.wiktionary's Urdu entries and ur.wiktionary's English entries. Data: CC BY-SA, Wiktionary contributors.
import { spawn } from 'node:child_process';
import { createInterface } from 'node:readline';
import { Readable } from 'node:stream';
import { pool } from './db.ts';
export const LANGS = ['ur', 'fa', 'ar'] as const;
const NAME = { ur: 'Urdu', fa: 'Persian', ar: 'Arabic' } as const;
const UA = 'Divan/2 (https://github.com/anas-rashid/divan)';
const LANGS = { ur: 'Urdu', fa: 'Persian', ar: 'Arabic' } as const;
const DIAC = /[ً-ْٰٔٗ٘]/g;
const plain = (html: string) =>
html.replace(/<[^>]+>/g, '').replace(/&nbsp;/g, ' ').replace(/&amp;/g, '&').replace(/&#\d+;/g, '').replace(/\s+/g, ' ').trim();
const KAIKKI = (l: string) => `https://kaikki.org/dictionary/${NAME[l as 'ur']}/kaikki.org-dictionary-${NAME[l as 'ur']}.jsonl`;
const DUMP = (l: string) => `https://dumps.wikimedia.org/${l}wiktionary/latest/${l}wiktionary-latest-pages-articles.xml.bz2`;
const MARKS = /[ؐ-ًؚ-ٰٟۖ-ۭـ‌‍‎‏]/g;
// spelling-insensitive key shared by Urdu, Persian and Arabic: no diacritics/tatweel, one heh, one yeh, one kaf,
// plain alef, noon ghunna as noon (ناداں = نادان, قسمت = قِسْمَت, نگاہ = نگاه, معنی = معنى). ے and ھ stay distinct.
export const key = (w: string) =>
w.normalize('NFC').replace(MARKS, '')
.replace(/[ہۂۃةه]/g, 'ه').replace(/[يىی]/g, 'ی')
.replace(/ك/g, 'ک').replace(/[أإٱ]/g, 'ا').replace(/ں/g, 'ن').trim();
const unlink = (wt: string) => wt.replace(/\[\[(?:[^\]|]*\|)?([^\]]*)\]\]/g, '$1').replace(/'''?/g, '').trim();
const upload = (u?: string) => (u && u.startsWith('https://upload.wikimedia.org/') ? u : null);
async function get(url: string, json = true): Promise<any> {
const res = await fetch(url, { headers: { 'user-agent': UA }, signal: AbortSignal.timeout(8000) }).catch(() => null);
if (!res?.ok) return null;
return json ? res.json() : res.text();
}
const rest = (path: string, title: string) => `https://en.wiktionary.org/api/rest_v1/page/${path}/${encodeURIComponent(title)}`;
const wikitext = async (host: string, title: string): Promise<string> => {
const d = await get(`https://${host}/w/api.php?action=parse&page=${encodeURIComponent(title)}&prop=wikitext&format=json&formatversion=2&redirects=1`);
return d?.parse?.wikitext ?? '';
};
// ---- parsing ----
// spelling forms to try: as written, without diacritics, final noon ghunna as noon (ناداں -> نادان)
export const forms = (w: string) => {
const bare = w.replace(DIAC, '').replace(/ۂ/g, 'ہ');
return [...new Set([w, bare, bare.replace(/ں$/, 'ن')])];
};
// the word in Persian / Arabic spelling (ہ -> ه, ی -> ي, ک -> ك …); Arabic also tries final alef maqsura
const SPELL: Record<string, [RegExp, string][]> = {
fa: [[/[\u06C1\u06C2\u06BE]/g, '\u0647'], [/[\u06D2\u064A]/g, '\u06CC'], [/\u0643/g, '\u06A9']],
ar: [[/[\u06C1\u06BE]/g, '\u0647'], [/\u06C2/g, '\u0629'], [/[\u06CC\u06D2]/g, '\u064A'], [/\u06A9/g, '\u0643']],
};
export function formsIn(code: string, w: string) {
const base = forms(w);
if (!SPELL[code]) return base;
const conv = base.map((f) => SPELL[code].reduce((s, [re, to]) => s.replace(re, to), f));
return [...new Set([...conv, ...(code === 'ar' ? conv.map((f) => f.replace(/\u064A$/, '\u0649')) : [])])];
// a kaikki.org (Wiktextract) line -> the fields the sidebar shows
export function kaikkiEntry(d: any) {
const senses = (d.senses ?? []).filter((s: any) => s.glosses?.length);
const sounds = d.sounds ?? [];
return {
pos: d.pos as string,
glosses: senses.map((s: any) => s.glosses.join('; ')).slice(0, 6) as string[],
formOf: senses.length > 0 && senses.every((s: any) => s.form_of || s.tags?.includes('form-of')),
ipa: [...new Set(sounds.map((s: any) => s.ipa).filter(Boolean))].slice(0, 2) as string[],
audio: sounds.map((s: any) => upload(s.mp3_url) ?? upload(s.ogg_url)).find(Boolean) ?? null,
tr: d.forms?.find((f: any) => f.tags?.includes('romanization'))?.form ?? null,
ety: d.etymology_text ? etymology(String(d.etymology_text)) : null,
synonyms: (d.synonyms ?? []).map((s: any) => s.word).filter(Boolean).slice(0, 8) as string[],
};
}
// fa./ar.wiktionary: "# definition" lines outside etymology/translation sections; on ar.wiktionary only
// the Arabic section, and undiacritised pages that just point to entries ("* [[عِشْق]]") give those links
// etymology text without Wiktionary's "Etymology tree …" diagram summary; cut at 300 characters, never
// leaving half a surrogate pair
const etymology = (t: string) =>
(t.startsWith('Etymology tree') ? t.slice(Math.max(0, t.search(/\b(Borrowed|Inherited|From|Learned|Derived|Semi-learned|Calque|Compound)\b/))) : t)
.slice(0, 300).toWellFormed();
// English glosses of an Urdu entry as pivot keys: "fate, destiny" -> fate, destiny (short, lowercase)
export const glossKeys = (glosses: string[]) => [...new Set(glosses.flatMap((g) => g.replace(/\([^)]*\)/g, '').split(/[,;]/))
.map((s) => s.trim().toLowerCase().replace(/^(to|a|an|the) /, '')).filter((s) => /^[a-z][a-z' -]{1,30}$/.test(s) && s.split(' ').length <= 3))];
// ur.wiktionary Urdu entries: numbered lines under ==معانی==, else "# " lines, else the opening prose;
// origin from "(عربی)" on the first line
export function urduEntry(wt: string) {
const m = wt.split(/==\s*معانی\s*==/)[1]?.split(/\n==/)[0] ?? '';
let defs = m.split('\n').map((l) => l.match(/^\s*(?:\d+[.)-]|#)\s*(.+)/)?.[1]).filter(Boolean).map((l) => unlink(l!));
if (!defs.length) defs = definitions(wt, 'ur').defs;
if (!defs.length) {
const prose = wt.split('\n').find((l) => l.trim().startsWith("'''") && l.length > 20);
if (prose) defs = [unlink(stripTemplates(prose)).slice(0, 300)];
}
const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null;
const english = wt.match(/^\s*انگریزی\s*:\s*([A-Za-z].+)$/m)?.[1].trim() ?? null; // "== تراجم ==" pages
return { defs: defs.slice(0, 6), origin, links: [] as string[], ...(english && { english }) };
}
// ur.wiktionary English entries ("north": "# [[شمالی]]۔"): the Urdu words, for the pivot through English
export const isEnglish = (title: string) => /^[A-Za-z][A-Za-z' -]*$/.test(title);
export function englishToUrdu(wt: string) {
const lines = wt.split('\n').filter((l) => /^#(?![:*])/.test(l)).join(' ');
return [...new Set([...lines.matchAll(/\[\[(?:[^\]|]*\|)?([^\]]+)\]\]/g)].map((m) => m[1].trim())
.filter((w) => /^[\p{Script=Arabic}\s]+$/u.test(w)))].slice(0, 10);
}
const stripTemplates = (t: string) => {
while (/\{\{[^{}]*\}\}/.test(t)) t = t.replace(/\{\{[^{}]*\}\}/g, '');
return t;
};
// fa./ar.wiktionary: "# definition" lines outside etymology/translation sections; on ar.wiktionary only the
// Arabic section, and undiacritised pages that just point to entries ("* [[عِشْق]]") give those links
export function definitions(wt: string, code: string) {
if (code === 'ar' && wt.includes('{{اللغة|')) wt = wt.split(/==\s*\{\{اللغة\|/).find((x) => x.startsWith('عربية')) ?? '';
const defs: string[] = [], links: string[] = [];
@ -55,102 +96,214 @@ export function definitions(wt: string, code: string) {
if (only) { links.push(only[1]); continue; }
const m = line.match(/^#(?![:*])\s*(.+)/);
if (!m) continue;
let t = m[1];
while (/\{\{[^{}]*\}\}/.test(t)) t = t.replace(/\{\{[^{}]*\}\}/g, '');
t = unlink(t).replace(/^[\s.،:-]+|[\s]+$/g, '');
const t = unlink(stripTemplates(m[1])).replace(/^[\s.،:-]+|\s+$/g, '');
if (t.length >= 3 && !/^-+$/.test(t)) defs.push(t);
}
return { defs: defs.slice(0, 5), links };
return { defs: defs.slice(0, 6), origin: null as string | null, links: links.slice(0, 5) };
}
// meanings from the language's own Wiktionary: the first spelling with an entry (following one pointer hop)
async function native(code: string, word: string) {
const host = `${code}.wiktionary.org`;
for (const f of formsIn(code, word)) {
const wt = await wikitext(host, f);
if (!wt) continue;
if (code === 'ur') {
const e = urduEntry(wt);
if (e.meanings.length) return { defs: e.meanings, origin: e.origin, url: `https://${host}/wiki/${encodeURIComponent(f)}` };
continue;
export const ownEntry = (code: string, wt: string) => (code === 'ur' ? urduEntry(wt) : definitions(wt, code));
// pages of a MediaWiki XML dump (articles only, no redirects)
export function* dumpPages(xml: string) {
for (const m of xml.matchAll(/<page>([\s\S]*?)<\/page>/g)) {
const p = m[1];
if (!/<ns>0<\/ns>/.test(p) || /<redirect /.test(p)) continue;
const title = p.match(/<title>([^<]*)<\/title>/)?.[1];
const text = p.match(/<text[^>]*>([\s\S]*?)<\/text>/)?.[1];
if (title && text) yield { title: xmlText(title), text: xmlText(text) };
}
}
const xmlText = (s: string) => s.replace(/&lt;/g, '<').replace(/&gt;/g, '>').replace(/&quot;/g, '"').replace(/&#039;/g, "'").replace(/&amp;/g, '&');
// ---- import and sync ----
const meta = async (k: string) => (await pool.query('SELECT value FROM dict_meta WHERE name = $1', [k])).rows[0]?.value ?? null;
const setMeta = (k: string, v: string) =>
pool.query('INSERT INTO dict_meta (name, value) VALUES ($1, $2) ON CONFLICT (name) DO UPDATE SET value = $2', [k, v]);
async function insertRows(client: any, rows: unknown[][]) {
for (let i = 0; i < rows.length; i += 500) {
const chunk = rows.slice(i, i + 500), values: unknown[] = [];
const sql = chunk.map((r, j) => { values.push(...r); const o = j * 5; return `($${o + 1},$${o + 2},$${o + 3},$${o + 4},$${o + 5})`; });
await client.query(`INSERT INTO wiktionary (lang, source, title, key, data) VALUES ${sql.join(',')}`, values);
}
}
// replace one source's rows in a transaction (readers see the old data until it commits)
async function replace(lang: string, source: string, fill: (add: (title: string, data: object) => Promise<void>, client: any) => Promise<void>) {
const client = await pool.connect();
let n = 0;
try {
await client.query('BEGIN');
await client.query('DELETE FROM wiktionary WHERE lang = $1 AND source = $2', [lang, source]);
if (lang === 'ur') await client.query('DELETE FROM ur_glosses WHERE source = $1', [source]);
let batch: unknown[][] = [];
await fill(async (title, data) => {
batch.push([lang, source, title, key(title), data]); n++;
if (batch.length >= 2000) { await insertRows(client, batch); batch = []; }
}, client);
await insertRows(client, batch);
await client.query('COMMIT');
} catch (e) {
await client.query('ROLLBACK');
throw e;
} finally {
client.release();
}
return n;
}
async function lastModified(url: string) {
const res = await fetch(url, { method: 'HEAD', headers: { 'user-agent': UA } });
if (!res.ok) throw new Error(`${res.status} ${url}`);
return res.headers.get('last-modified') ?? '';
}
async function importKaikki(lang: string) {
const res = await fetch(KAIKKI(lang), { headers: { 'user-agent': UA } });
if (!res.ok || !res.body) throw new Error(`${res.status} ${KAIKKI(lang)}`);
return replace(lang, 'en', async (add, client) => {
const glossRows: string[][] = [];
for await (const line of createInterface({ input: Readable.fromWeb(res.body as any), crlfDelay: Infinity })) {
if (!line) continue;
const d = JSON.parse(line), e = kaikkiEntry(d);
if (!e.glosses.length && !e.ipa.length) continue;
await add(d.word, e);
if (lang === 'ur' && !e.formOf) for (const g of glossKeys(e.glosses.slice(0, 3))) glossRows.push([g, d.word]);
}
let { defs, links } = definitions(wt, code), title = f;
if (!defs.length && links[0]) ({ defs } = definitions(await wikitext(host, (title = links[0])), code));
if (defs.length) return { defs, origin: null, url: `https://${host}/wiki/${encodeURIComponent(title)}` };
await insertGlosses(client, 'en', glossRows);
});
}
async function insertGlosses(client: any, source: string, rows: string[][]) {
for (let i = 0; i < rows.length; i += 1000) {
const chunk = rows.slice(i, i + 1000);
await client.query(`INSERT INTO ur_glosses (gloss, word, source) VALUES ${chunk.map((_, j) => `($${j * 2 + 1},$${j * 2 + 2},'${source}')`).join(',')}`, chunk.flat());
}
return null;
}
// ur.wiktionary: numbered lines under ==معانی==, origin from "(عربی)" on the first line
export function urduEntry(wt: string) {
const m = wt.split(/==\s*معانی\s*==/)[1]?.split(/\n==/)[0] ?? '';
const meanings = m.split('\n').map((l) => l.match(/^\s*(?:\d+[.)-]|#)\s*(.+)/)?.[1]).filter(Boolean).map((l) => unlink(l!)).slice(0, 6);
const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null;
return { meanings, origin };
async function importDump(lang: string) {
const bz = spawn('sh', ['-c', `curl -sfL -A '${UA}' '${DUMP(lang)}' | bzcat`]);
return replace(lang, 'own', async (add, client) => {
let buf = '';
const glossRows: string[][] = []; // ur.wiktionary English entries -> the pivot table
for await (const chunk of bz.stdout.setEncoding('utf8')) {
buf += chunk;
const end = buf.lastIndexOf('</page>');
if (end < 0) continue;
for (const p of dumpPages(buf.slice(0, end + 7))) {
if (lang === 'ur' && isEnglish(p.title)) {
for (const w of englishToUrdu(p.text)) glossRows.push([p.title.toLowerCase(), w]);
continue;
}
const e = ownEntry(lang, p.text);
if (e.defs.length || e.links.length || 'english' in e) await add(p.title, e);
}
buf = buf.slice(end + 7);
}
const code: number = await new Promise((r) => (bz.exitCode !== null ? r(bz.exitCode) : bz.on('close', r)));
if (code !== 0) throw new Error(`dump download/decompress failed for ${lang} (exit ${code})`);
await insertGlosses(client, 'own', glossRows);
});
}
// en.wiktionary page HTML: per language, IPA, headword transliteration, audio
export function pronunciations(html: string) {
const out: Record<string, { ipa: string[]; tr: string | null; audio: string | null }> = {};
const parts = html.split(/<h2[^>]*id="([^"]+)"/);
for (let i = 1; i < parts.length; i += 2) {
const code = Object.entries(LANGS).find(([, n]) => n === parts[i])?.[0];
if (!code) continue;
const body = parts[i + 1];
const ipa = [...new Set([...body.matchAll(/class="IPA[^"]*"[^>]*>([^<]+)/g)].map((m) => m[1]).filter((x) => /^[/[]/.test(x)))].slice(0, 2);
const tr = body.match(new RegExp(`lang="${code}-Latn" class="headword-tr[^"]*"[^>]*>([^<]+)`))?.[1] ?? null;
const a = body.match(/"(?:https:)?(\/\/upload\.wikimedia\.org\/[^"]+\.(?:ogg|oga|mp3|wav))"/)?.[1];
out[code] = { ipa, tr: tr && plain(tr), audio: a ? 'https:' + a : null };
// changes on ur./fa./ar.wiktionary since the last run, re-read from the API (titles in batches of 50)
async function recentChanges(lang: string) {
const api = `https://${lang}.wiktionary.org/w/api.php`;
const since = await meta(`rc:${lang}`), now = new Date().toISOString();
if (!since) return setMeta(`rc:${lang}`, now).then(() => 0); // first run: the dump is the baseline
const titles = new Set<string>();
let cont: Record<string, string> = {};
do {
const q = new URLSearchParams({ action: 'query', list: 'recentchanges', rcnamespace: '0', rctype: 'edit|new', rcprop: 'title',
rclimit: '500', rcdir: 'newer', rcstart: since, rcend: now, format: 'json', formatversion: '2', ...cont });
const d: any = await (await fetch(`${api}?${q}`, { headers: { 'user-agent': UA } })).json();
for (const c of d.query?.recentchanges ?? []) titles.add(c.title);
cont = d.continue ?? {};
} while (cont.rccontinue);
const list = [...titles];
for (let i = 0; i < list.length; i += 50) {
const q = new URLSearchParams({ action: 'query', prop: 'revisions', rvprop: 'content', rvslots: 'main',
titles: list.slice(i, i + 50).join('|'), format: 'json', formatversion: '2' });
const d: any = await (await fetch(api, { method: 'POST', body: q, headers: { 'user-agent': UA } })).json();
for (const p of d.query?.pages ?? []) {
const wt = p.revisions?.[0]?.slots?.main?.content;
if (lang === 'ur' && isEnglish(p.title)) {
const g = p.title.toLowerCase();
await pool.query(`DELETE FROM ur_glosses WHERE source = 'own' AND gloss = $1`, [g]);
for (const w of wt ? englishToUrdu(wt) : []) await pool.query(`INSERT INTO ur_glosses (gloss, word, source) VALUES ($1, $2, 'own')`, [g, w]);
continue;
}
await pool.query(`DELETE FROM wiktionary WHERE lang = $1 AND source = 'own' AND title = $2`, [lang, p.title]);
const e = wt && !/^#(REDIRECT|تحويل|تغییر)/i.test(wt) ? ownEntry(lang, wt) : null;
if (e && (e.defs.length || e.links.length || 'english' in e))
await pool.query(`INSERT INTO wiktionary (lang, source, title, key, data) VALUES ($1, 'own', $2, $3, $4)`, [lang, p.title, key(p.title), e]);
}
}
return out;
await setMeta(`rc:${lang}`, now);
return list.length;
}
// Urdu equivalents of single-word English glosses, from their translation tables
async function pivot(word: string, glosses: string[]) {
const seen = new Set([word.replace(DIAC, '')]); // the word itself, and duplicates differing only in diacritics
const words = [...new Set(glosses.flatMap((g) => g.split(/[,;]/)).map((s) => s.trim()).filter((s) => /^[a-z]+$/.test(s)))].slice(0, 3); // lowercase: no proper nouns
const found = await Promise.all(words.map(async (gloss) => {
const wt = await wikitext('en.wiktionary.org', gloss);
const lines = [...wt.matchAll(/^\*:? Urdu: (.*)$/gm)].map((m) => m[1]).join(' ');
const urdu = [...lines.matchAll(/\{\{t\+?\|ur\|([^|}]+)/g)].map((m) => m[1].trim())
.filter((u) => !seen.has(u.replace(DIAC, '')) && seen.add(u.replace(DIAC, ''))).slice(0, 8);
return { gloss, urdu };
}));
return found.filter((f) => f.urdu.length);
// full re-import of each source whose upstream file changed, then the daily recent changes
export async function sync(log = console.log) {
for (const lang of LANGS) {
for (const [source, url, run] of [['en', KAIKKI(lang), importKaikki], ['own', DUMP(lang), importDump]] as const) {
const lm = await lastModified(url), k = `file:${lang}:${source}`;
if (lm && lm === (await meta(k))) continue;
const t = Date.now(), n = await run(lang);
await setMeta(k, lm);
if (source === 'own') await setMeta(`rc:${lang}`, new Date(lm).toISOString()); // changes after the dump
log(`${lang} ${source}: ${n} entries (${Math.round((Date.now() - t) / 1000)} s)`);
}
log(`${lang} recent changes: ${await recentChanges(lang)} pages`);
}
}
async function fetchWord(word: string) {
let title: string | null = null, defs: any = null;
for (const f of forms(word)) if ((defs = await get(rest('definition', f)))) { title = f; break; }
const [pron, ...own] = await Promise.all([
title ? get(rest('html', title), false).then((h) => pronunciations(h ?? '')) : {},
...Object.keys(LANGS).map((c) => native(c, word)),
]) as [Record<string, any>, ...(Awaited<ReturnType<typeof native>>)[]];
const langs = Object.keys(LANGS).map((code, i) => ({
code, ...(pron[code] ?? { ipa: [], tr: null, audio: null }),
meanings: own[i]?.defs ?? [], origin: own[i]?.origin ?? null, source: own[i]?.url ?? null,
senses: (defs?.[code] ?? []).map((e: any) => ({
pos: e.partOfSpeech, defs: e.definitions.map((d: any) => plain(d.definition)).filter(Boolean).slice(0, 4),
})).filter((s: any) => s.defs.length).slice(0, 3),
})).filter((l) => l.meanings.length || l.senses.length || l.ipa.length || l.tr);
const glosses = langs.find((l) => l.senses.length)?.senses[0].defs.slice(0, 2) ?? []; // the primary meaning only
const hasUrdu = langs.some((l) => l.code === 'ur' && l.meanings.length);
return {
word, langs,
equivalents: hasUrdu ? [] : await pivot(word, glosses),
sources: { en: title && `https://en.wiktionary.org/wiki/${encodeURIComponent(title)}` },
found: langs.length > 0,
};
}
// ---- lookup ----
const page = (lang: string, source: string, title: string) =>
source === 'en' ? `https://en.wiktionary.org/wiki/${encodeURIComponent(title)}#${NAME[lang as 'ur']}` : `https://${lang}.wiktionary.org/wiki/${encodeURIComponent(title)}`;
export async function lookup(word: string) {
const hit = await pool.query(`SELECT data FROM dictionary WHERE word = $1 AND fetched_at > now() - interval '30 days'`, [word]);
if (hit.rows[0]) return hit.rows[0].data;
const data = await fetchWord(word);
if (!data.found) return data; // not found (or Wiktionary unreachable): don't cache
await pool.query(
`INSERT INTO dictionary (word, data) VALUES ($1, $2) ON CONFLICT (word) DO UPDATE SET data = $2, fetched_at = now()`,
[word, data],
);
return data;
const k = key(word);
const { rows } = await pool.query('SELECT lang, source, title, data FROM wiktionary WHERE key = $1 ORDER BY id', [k]);
// ar.wiktionary pointer pages: follow to the diacritised entries
const pointers = rows.filter((r) => r.source === 'own' && !r.data.defs.length).flatMap((r) => r.data.links.map((l: string) => [r.lang, l]));
if (pointers.length) {
const more = await pool.query(
`SELECT lang, source, title, data FROM wiktionary WHERE source = 'own' AND (lang, title) IN (SELECT * FROM unnest($1::text[], $2::text[]))`,
[pointers.map((p) => p[0]), pointers.map((p) => p[1])]);
rows.push(...more.rows);
}
const langs = LANGS.map((code) => {
const en = rows.filter((r) => r.lang === code && r.source === 'en');
const own = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.defs.length);
const lemmas = en.filter((r) => !r.data.formOf), use = lemmas.length ? lemmas : en;
const first = (f: string) => use.map((r) => r.data[f]).find((v) => (Array.isArray(v) ? v.length : v)) ?? null;
// ur.wiktionary's English translation lines ("انگریزی : …")
const translated = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.english).map((r) => r.data.english);
return {
code, ipa: first('ipa') ?? [], tr: first('tr'), audio: first('audio'), ety: first('ety'),
synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))].slice(0, 8),
meanings: own.flatMap((r) => r.data.defs).slice(0, 6), origin: own.find((r) => r.data.origin)?.data.origin ?? null,
source: own[0] ? page(code, 'own', own[0].title) : null,
senses: [...use.map((r) => ({ pos: r.data.pos, defs: r.data.glosses.slice(0, 4) })).filter((s) => s.defs.length).slice(0, 3),
...(translated.length ? [{ pos: 'translation', defs: translated }] : [])],
en: en[0] ? page(code, 'en', en[0].title) : null,
};
}).filter((l) => l.meanings.length || l.senses.length || l.ipa.length || l.tr);
// no Urdu meaning: Urdu words sharing the primary English meaning
let equivalents: { gloss: string; urdu: string[] }[] = [];
if (!langs.some((l) => l.code === 'ur' && l.meanings.length)) {
const glosses = glossKeys(langs.find((l) => l.senses.length)?.senses[0].defs.slice(0, 2) ?? []).slice(0, 4);
if (glosses.length) {
const { rows: eq } = await pool.query('SELECT gloss, word FROM ur_glosses WHERE gloss = ANY($1)', [glosses]);
const seen = new Set([k]);
equivalents = glosses.map((g) => ({
gloss: g, urdu: eq.filter((r) => r.gloss === g && !seen.has(key(r.word)) && seen.add(key(r.word))).map((r) => r.word).slice(0, 8),
})).filter((e) => e.urdu.length);
}
}
return { word, langs, equivalents, found: langs.length > 0 };
}

View File

@ -58,9 +58,25 @@ CREATE TABLE IF NOT EXISTS verses (
PRIMARY KEY (poem_id, vorder)
);
-- Wiktionary lookups for the reading sidebar (api/src/dictionary.ts), refreshed after 30 days
CREATE TABLE IF NOT EXISTS dictionary (
word text PRIMARY KEY,
data jsonb NOT NULL,
fetched_at timestamptz NOT NULL DEFAULT now()
-- Wiktionary for the word sidebar (api/src/dictionary.ts; filled and kept current by dict-sync.ts)
DROP TABLE IF EXISTS dictionary; -- the earlier live-lookup cache
CREATE TABLE IF NOT EXISTS wiktionary (
id bigserial PRIMARY KEY,
lang text NOT NULL, -- the word's language: ur | fa | ar
source text NOT NULL, -- en (en.wiktionary, via kaikki.org) | own (that language's Wiktionary)
title text NOT NULL, -- headword / page title
key text NOT NULL, -- spelling-insensitive lookup key (dictionary.ts key())
data jsonb NOT NULL
);
CREATE INDEX IF NOT EXISTS wiktionary_key ON wiktionary(key);
CREATE INDEX IF NOT EXISTS wiktionary_title ON wiktionary(lang, source, title);
CREATE TABLE IF NOT EXISTS ur_glosses ( -- English gloss -> Urdu word, for the pivot through English
gloss text NOT NULL,
word text NOT NULL
);
ALTER TABLE ur_glosses ADD COLUMN IF NOT EXISTS source text NOT NULL DEFAULT 'en'; -- en | own (ur.wiktionary English entries)
CREATE INDEX IF NOT EXISTS ur_glosses_gloss ON ur_glosses(gloss);
CREATE TABLE IF NOT EXISTS dict_meta ( -- upstream file versions and recent-changes timestamps
name text PRIMARY KEY,
value text NOT NULL
);

View File

@ -4,6 +4,7 @@
# 1. update the divan-data checkout and fetch new/edited works from Wikisource (incremental)
# 2. rebuild its search index and site export
# 3. load the export into PostgreSQL (upserts; the site reads it live)
# 3b. sync the word dictionary from Wiktionary (api/src/dict-sync.ts; first run imports ~700 MB)
# 4. optionally commit + push the refreshed data (DIVAN_DATA_PUSH=1, needs git push access)
#
# Schedule with cron, e.g. daily at 03:15:
@ -11,7 +12,7 @@
#
# Settings (env): DIVAN_DATA_DIR (default /opt/divan-data), DIVAN_APP_DIR (default: this repo),
# DATABASE_URL (Postgres, as for the API), DIVAN_DATA_PUSH=1 to push data changes.
# Needs: git, python3 (stdlib only), node 24+.
# Needs: git, python3 (stdlib only), node 24+, curl and bzcat (dictionary dumps).
set -euo pipefail
APP_DIR=${DIVAN_APP_DIR:-$(cd "$(dirname "$0")/.." && pwd)}
@ -36,4 +37,8 @@ fi
cd "$APP_DIR/api"
[ -d node_modules ] || npm ci --omit=dev --silent
node src/import.ts "$DATA_DIR"
# word dictionary (Wiktionary): re-imports sources that changed upstream, then the day's edits.
# A failure here must not fail the content sync.
node src/dict-sync.ts || echo "$(date -u +%FT%TZ) dictionary sync failed"
echo "== $(date -u +%FT%TZ) done"

View File

@ -120,7 +120,8 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
const dict = document.getElementById('dict'), dictBody = dict.querySelector('.dict-body'), main = document.querySelector('main');
const LANG = { ur: 'اردو', fa: 'فارسی', ar: 'عربی' };
const POS = { Noun: 'اسم', 'Proper noun': 'اسم معرفہ', Verb: 'فعل', Adjective: 'صفت', Adverb: 'متعلق فعل', Pronoun: 'ضمیر',
Postposition: 'حرف جار', Preposition: 'حرف جار', Conjunction: 'حرف عطف', Interjection: 'حرف ندا', Particle: 'حرف' };
Postposition: 'حرف جار', Preposition: 'حرف جار', Conjunction: 'حرف عطف', Interjection: 'حرف ندا', Particle: 'حرف',
Translation: 'ترجمہ (اردو ویکی لغت)', Phrase: 'فقرہ', Proverb: 'کہاوت', Name: 'اسم معرفہ' };
const el = (tag, cls, text, attrs) => {
const e = document.createElement(tag);
if (cls) e.className = cls;
@ -134,7 +135,8 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
current = w; dict.hidden = false; document.body.classList.add('dict-open');
dictBody.replaceChildren(el('h2', null, w), el('p', 'muted', 'تلاش جاری ہے…'));
let d = null;
try { const r = await fetch('/api/word?w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {}
// v: bump when the response format changes (responses are browser-cached for a day)
try { const r = await fetch('/api/word?v=2&w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {}
if (w !== current) return; // a newer selection won
const out = [el('h2', null, w)];
if (!d || !d.found) {
@ -171,14 +173,20 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
if (l.source) label.append(el('a', null, 'ماخذ', { href: l.source, target: '_blank', rel: 'noopener' }));
sec.append(label, list(l.meanings, null, { lang: l.code }));
}
for (const s of l.senses) sec.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' }));
for (const s of l.senses) sec.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos.charAt(0).toUpperCase() + s.pos.slice(1)] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' }));
if (l.synonyms.length) sec.append(el('p', 'pos', 'مترادفات'), el('p', null, l.synonyms.join('، ')));
if (l.ety) sec.append(el('p', 'pos', 'اشتقاق'), el('p', 'en ety', l.ety, { dir: 'ltr', lang: 'en' }));
if (l.en) {
const p = el('p', 'pos');
p.append(el('a', null, 'انگریزی ویکی لغت', { href: l.en, target: '_blank', rel: 'noopener' }));
sec.append(p);
}
out.push(sec);
if (l.code === 'ur' && d.equivalents.length) out.push(equivalents());
}
if (d.equivalents.length && !d.langs.some((l) => l.code === 'ur')) out.splice(1, 0, equivalents());
const src = el('p', 'src muted');
src.append('ویکی لغت · ');
if (d.sources.en) src.append(el('a', null, 'انگریزی ویکی لغت', { href: d.sources.en, target: '_blank', rel: 'noopener' }), ' · ');
src.append('ویکی لغت (Wiktionary) · ');
src.append('CC BY-SA');
dictBody.replaceChildren(...out, src);
}

View File

@ -146,6 +146,7 @@ h1 + .muted { text-align: center; margin-top: 0; }
.dict .pos { font-size: .8rem; color: var(--muted); margin: 8px 0 0; }
.dict .note { font-size: .8rem; margin: 0; }
.dict .src { font-size: .78rem; margin-top: 18px; }
.dict .ety { text-align: left; color: var(--muted); font-size: .8rem; }
.dict-close { position: sticky; top: 0; float: left; border: 0; background: none; color: var(--muted); font-size: 1rem; cursor: pointer; }
.dict .play { border: 1px solid var(--border); background: var(--inner); border-radius: 8px; cursor: pointer; padding: 0 6px; }
/* wide screens: the page moves over instead of being covered */