diff --git a/README.md b/README.md
index 9208d68a..83a5a022 100644
--- a/README.md
+++ b/README.md
@@ -42,6 +42,8 @@ The import upserts, so re-running it after a divan-data sync applies the changes
15 3 * * * DATABASE_URL=postgres://divan:...@localhost:5432/divan /opt/divan/deploy/sync.sh >> /var/log/divan-sync.log 2>&1
```
+The same script keeps the word dictionary current (`npm run dict-sync` in `api/`): the full Wiktionary data for Urdu, Persian and Arabic (English Wiktionary via [kaikki.org](https://kaikki.org), and the Urdu, Persian and Arabic Wiktionary dumps) is re-imported when upstream publishes new files, and each day's Wiktionary edits are applied from recent changes. The first run downloads about 700 MB.
+
Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DIVAN_APP_DIR` (default: this repo), `DIVAN_DATA_PUSH=1` to also commit and push data changes (needs git push access). Needs git, python3 and Node 24+.
## API
@@ -51,7 +53,7 @@ Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DI
| `GET /api/poets` | all poets |
| `GET /api/page?url=/p238/...` | the poet, category or poem at a site URL (breadcrumbs, children, verses, prev/next) |
| `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised; exact phrase first; each with the best-matching couplet or paragraph (`snippet`) |
-| `GET /api/word?w=` | one word's meanings and pronunciation from Wiktionary (Urdu, Persian, Arabic in that order; English meanings; Urdu equivalents via English when Urdu Wiktionary has none), cached 30 days |
+| `GET /api/word?w=` | one word's meanings and pronunciation from the local Wiktionary data (Urdu, Persian, Arabic in that order; English meanings; Urdu equivalents via English when Urdu Wiktionary has none) |
| `GET /health` | database check |
Search normalises both stored text and queries: Arabic ي/ك/ه → Urdu ی/ک/ہ, ۂ/ۓ, diacritics and the Urdu full stop removed; do-chashmi ھ stays distinct.
diff --git a/api/package.json b/api/package.json
index 4c7a48b7..c5206c0f 100644
--- a/api/package.json
+++ b/api/package.json
@@ -9,7 +9,8 @@
"scripts": {
"start": "node src/server.ts",
"import": "node src/import.ts",
- "test": "node --test src/*.test.ts"
+ "test": "node --test src/*.test.ts",
+ "dict-sync": "node src/dict-sync.ts"
},
"dependencies": {
"fastify": "^5.12.5",
diff --git a/api/src/dict-sync.ts b/api/src/dict-sync.ts
new file mode 100644
index 00000000..494fada0
--- /dev/null
+++ b/api/src/dict-sync.ts
@@ -0,0 +1,10 @@
+// Import / sync the Wiktionary data (see dictionary.ts). First run imports everything (about 1 GB of downloads);
+// later runs re-import only sources that changed upstream and apply the day's Wiktionary edits.
+// node src/dict-sync.ts
+import { readFile } from 'node:fs/promises';
+import { pool } from './db.ts';
+import { sync } from './dictionary.ts';
+
+await pool.query(await readFile(new URL('../../db/schema.sql', import.meta.url), 'utf8'));
+await sync((line) => console.log(`${new Date().toISOString()} ${line}`));
+await pool.end();
diff --git a/api/src/dictionary.test.ts b/api/src/dictionary.test.ts
index 4036aa44..6526eda8 100644
--- a/api/src/dictionary.test.ts
+++ b/api/src/dictionary.test.ts
@@ -1,34 +1,67 @@
import { test } from 'node:test';
import assert from 'node:assert/strict';
-import { forms, urduEntry, pronunciations } from './dictionary.ts';
+import { key, kaikkiEntry, glossKeys, urduEntry, definitions, dumpPages } from './dictionary.ts';
-test('spelling forms: diacritics dropped, final noon ghunna tried as noon', () => {
- assert.deepEqual(forms('ناداں'), ['ناداں', 'نادان']);
- assert.deepEqual(forms('وِصال'), ['وِصال', 'وصال']);
+test('lookup key: spelling variants across Urdu, Persian and Arabic meet', () => {
+ const same = (a: string, b: string) => assert.equal(key(a), key(b), `${a} vs ${b}`);
+ same('ناداں', 'نادان');
+ same('قِسْمَت', 'قسمت');
+ same('نگاہ', 'نگاه');
+ same('معنی', 'معنى');
+ same('كتاب', 'کتاب');
+ assert.notEqual(key('کھ'), key('کہ'), 'do-chashmi heh stays distinct');
+ assert.notEqual(key('دیوانے'), key('دیوانی'), 'bari ye stays distinct');
+});
+
+test('kaikki entry: glosses, IPA, audio, romanisation, form-of', () => {
+ const e = kaikkiEntry({
+ word: 'قسمت', pos: 'noun', etymology_text: 'From Arabic',
+ senses: [{ glosses: ['fate, destiny'] }, { glosses: ['division'] }, { links: [] }],
+ sounds: [{ ipa: '/qɪs.mət̪/' }, { audio: 'x.wav', mp3_url: 'https://upload.wikimedia.org/x.mp3' }, { mp3_url: 'https://evil.example/x.mp3' }],
+ forms: [{ form: 'قِسْمَت', tags: ['canonical'] }, { form: 'qismat', tags: ['romanization'] }],
+ });
+ assert.deepEqual(e, { pos: 'noun', glosses: ['fate, destiny', 'division'], formOf: false, ipa: ['/qɪs.mət̪/'],
+ audio: 'https://upload.wikimedia.org/x.mp3', tr: 'qismat', ety: 'From Arabic', synonyms: [] });
+ assert.equal(kaikkiEntry({ word: 'x', pos: 'noun', senses: [{ glosses: ['plural of y'], tags: ['form-of'] }] }).formOf, true);
+});
+
+test('pivot keys from English glosses', () => {
+ assert.deepEqual(glossKeys(['fate, destiny', 'to love (someone)', 'a very long description of something that is not a key']), ['fate', 'destiny', 'love']);
});
test('ur.wiktionary entry: meanings and origin', () => {
- const wt = "وِصال {وِصال} ([[عربی]])\n\nاسم [[نکرہ]]\n\n==معانی==\n\n1. بھیٹ، ملاقات، [[عاشق و معشوق]] کی محبت۔\n\n==مرکبات==\n\nوِصال کا دِن";
- assert.deepEqual(urduEntry(wt), { meanings: ['بھیٹ، ملاقات، عاشق و معشوق کی محبت۔'], origin: 'عربی' });
+ const wt = 'وِصال {وِصال} ([[عربی]])\n\nاسم [[نکرہ]]\n\n==معانی==\n\n1. بھیٹ، ملاقات، [[عاشق و معشوق]] کی محبت۔\n\n==مرکبات==\n\nوِصال کا دِن';
+ assert.deepEqual(urduEntry(wt), { defs: ['بھیٹ، ملاقات، عاشق و معشوق کی محبت۔'], origin: 'عربی', links: [] });
});
-test('en.wiktionary HTML: IPA, transliteration and audio per language', () => {
- const html = '
Persian
/qis.ˈmat/-at' +
- 'qismat' +
- 'Urdu
/qɪs.mət̪/qismat';
- assert.deepEqual(pronunciations(html), {
- fa: { ipa: ['/qis.ˈmat/'], tr: 'qismat', audio: 'https://upload.wikimedia.org/a/b.ogg' },
- ur: { ipa: ['/qɪs.mət̪/'], tr: 'qismat', audio: null },
- });
-});
-
-test('fa./ar.wiktionary definitions: skip etymology, follow Arabic pointers', async () => {
- const { definitions, formsIn } = await import('./dictionary.ts');
+test('fa./ar.wiktionary definitions: skip etymology, Arabic pointers', () => {
const fa = '==فارسی==\n===ریشهشناسی===\n* [[عربی]]\n# قسمة\n===اسم===\n#بهره، نصیب.\n#:example\n#سرنوشت، تقدیر.\n====برگردانها====\n# x y z';
assert.deepEqual(definitions(fa, 'fa').defs, ['بهره، نصیب.', 'سرنوشت، تقدیر.']);
const ar = '== {{اللغة|عربية}} ==\nهل تقصد:\n* [[عِشْق]]\n* [[عَشَقَ]]\n----\n== {{اللغة|أردية}} ==\n# [[x]]';
- assert.deepEqual(definitions(ar, 'ar'), { defs: [], links: ['عِشْق', 'عَشَقَ'] });
+ assert.deepEqual(definitions(ar, 'ar'), { defs: [], origin: null, links: ['عِشْق', 'عَشَقَ'] });
assert.deepEqual(definitions('# {{مصدر|عَشِقَ}}.\n# فرط الحب.', 'ar').defs, ['فرط الحب.']);
- assert.deepEqual(formsIn('fa', 'نگاہ'), ['نگاه']);
- assert.deepEqual(formsIn('ar', 'معنی'), ['معني', 'معنى']);
+});
+
+test('dump pages: articles only, entities decoded, redirects skipped', () => {
+ const xml = 'عشق0a & b <x>' +
+ 'Talk:x1tr0#R';
+ assert.deepEqual([...dumpPages(xml)], [{ title: 'عشق', text: 'a & b ' }]);
+});
+
+test('etymology cut never leaves half a surrogate pair', () => {
+ const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'a'.repeat(299) + '𑀭𑀭' }).ety!;
+ assert.ok(ety.isWellFormed());
+});
+
+test('ur.wiktionary: English entries give Urdu words; Urdu entries fall back to # lines or prose', async () => {
+ const { englishToUrdu, isEnglish } = await import('./dictionary.ts');
+ assert.ok(isEnglish('north') && !isEnglish('شمال'));
+ assert.deepEqual(englishToUrdu('==انگریزی==\n===صفت===\n{{en-adjective}}\n# [[شمالی]]۔\n# [[شمال]] کی [[جانب]]۔\n#: [[x]]'), ['شمالی', 'شمال', 'جانب']);
+ assert.deepEqual(urduEntry('===اسم===\n# [[بہار]] کا [[موسم]]۔').defs, ['بہار کا موسم۔']);
+ assert.deepEqual(urduEntry("'''پھوڑی''' اس چادر کو کہتے ہیں جس پر بیٹھتے ہیں۔").defs, ['پھوڑی اس چادر کو کہتے ہیں جس پر بیٹھتے ہیں۔']);
+});
+
+test('etymology drops the "Etymology tree" summary', () => {
+ const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'Etymology tree Arabic قَسَمَ (qasama)bor. Urdu قِسْمَت Borrowed from Classical Persian قِسْمَت (qismat).' }).ety;
+ assert.equal(ety, 'Borrowed from Classical Persian قِسْمَت (qismat).');
});
diff --git a/api/src/dictionary.ts b/api/src/dictionary.ts
index e38240f1..2336dae5 100644
--- a/api/src/dictionary.ts
+++ b/api/src/dictionary.ts
@@ -1,48 +1,89 @@
-// Word lookup for the reading sidebar, from Wiktionary (CC BY-SA). For Urdu, Persian and Arabic, in that order:
-// - meanings in the language itself from its own Wiktionary (ur., fa., ar.wiktionary)
-// - English meanings and pronunciation (IPA, transliteration, audio) from en.wiktionary
-// With no Urdu meanings, Urdu equivalents are pivoted through English (the English glosses' translation tables).
-// Found words are cached in PostgreSQL (dictionary table) for 30 days.
+// Word dictionary for the reading sidebar: the full Wiktionary data for Urdu, Persian and Arabic, in PostgreSQL.
+// source 'en': en.wiktionary entries (English meanings, IPA, transliteration, audio, etymology, synonyms),
+// from the kaikki.org Wiktextract extracts (weekly)
+// source 'own': each language's own Wiktionary (ur., fa., ar.wiktionary): meanings in that language,
+// from the Wikimedia dumps (twice a month) plus recent changes (daily)
+// Urdu equivalents for words with no Urdu meaning are pivoted through English: ur_glosses maps English words to
+// Urdu words, from the English glosses of en.wiktionary's Urdu entries and ur.wiktionary's English entries. Data: CC BY-SA, Wiktionary contributors.
+import { spawn } from 'node:child_process';
+import { createInterface } from 'node:readline';
+import { Readable } from 'node:stream';
import { pool } from './db.ts';
+export const LANGS = ['ur', 'fa', 'ar'] as const;
+const NAME = { ur: 'Urdu', fa: 'Persian', ar: 'Arabic' } as const;
const UA = 'Divan/2 (https://github.com/anas-rashid/divan)';
-const LANGS = { ur: 'Urdu', fa: 'Persian', ar: 'Arabic' } as const;
-const DIAC = /[ً-ْٰٔٗ٘]/g;
-const plain = (html: string) =>
- html.replace(/<[^>]+>/g, '').replace(/ /g, ' ').replace(/&/g, '&').replace(/\d+;/g, '').replace(/\s+/g, ' ').trim();
+const KAIKKI = (l: string) => `https://kaikki.org/dictionary/${NAME[l as 'ur']}/kaikki.org-dictionary-${NAME[l as 'ur']}.jsonl`;
+const DUMP = (l: string) => `https://dumps.wikimedia.org/${l}wiktionary/latest/${l}wiktionary-latest-pages-articles.xml.bz2`;
+const MARKS = /[ؐ-ًؚ-ٰٟۖ-ۭـ]/g;
+
+// spelling-insensitive key shared by Urdu, Persian and Arabic: no diacritics/tatweel, one heh, one yeh, one kaf,
+// plain alef, noon ghunna as noon (ناداں = نادان, قسمت = قِسْمَت, نگاہ = نگاه, معنی = معنى). ے and ھ stay distinct.
+export const key = (w: string) =>
+ w.normalize('NFC').replace(MARKS, '')
+ .replace(/[ہۂۃةه]/g, 'ه').replace(/[يىی]/g, 'ی')
+ .replace(/ك/g, 'ک').replace(/[أإٱ]/g, 'ا').replace(/ں/g, 'ن').trim();
+
const unlink = (wt: string) => wt.replace(/\[\[(?:[^\]|]*\|)?([^\]]*)\]\]/g, '$1').replace(/'''?/g, '').trim();
+const upload = (u?: string) => (u && u.startsWith('https://upload.wikimedia.org/') ? u : null);
-async function get(url: string, json = true): Promise {
- const res = await fetch(url, { headers: { 'user-agent': UA }, signal: AbortSignal.timeout(8000) }).catch(() => null);
- if (!res?.ok) return null;
- return json ? res.json() : res.text();
-}
-const rest = (path: string, title: string) => `https://en.wiktionary.org/api/rest_v1/page/${path}/${encodeURIComponent(title)}`;
-const wikitext = async (host: string, title: string): Promise => {
- const d = await get(`https://${host}/w/api.php?action=parse&page=${encodeURIComponent(title)}&prop=wikitext&format=json&formatversion=2&redirects=1`);
- return d?.parse?.wikitext ?? '';
-};
+// ---- parsing ----
-// spelling forms to try: as written, without diacritics, final noon ghunna as noon (ناداں -> نادان)
-export const forms = (w: string) => {
- const bare = w.replace(DIAC, '').replace(/ۂ/g, 'ہ');
- return [...new Set([w, bare, bare.replace(/ں$/, 'ن')])];
-};
-
-// the word in Persian / Arabic spelling (ہ -> ه, ی -> ي, ک -> ك …); Arabic also tries final alef maqsura
-const SPELL: Record = {
- fa: [[/[\u06C1\u06C2\u06BE]/g, '\u0647'], [/[\u06D2\u064A]/g, '\u06CC'], [/\u0643/g, '\u06A9']],
- ar: [[/[\u06C1\u06BE]/g, '\u0647'], [/\u06C2/g, '\u0629'], [/[\u06CC\u06D2]/g, '\u064A'], [/\u06A9/g, '\u0643']],
-};
-export function formsIn(code: string, w: string) {
- const base = forms(w);
- if (!SPELL[code]) return base;
- const conv = base.map((f) => SPELL[code].reduce((s, [re, to]) => s.replace(re, to), f));
- return [...new Set([...conv, ...(code === 'ar' ? conv.map((f) => f.replace(/\u064A$/, '\u0649')) : [])])];
+// a kaikki.org (Wiktextract) line -> the fields the sidebar shows
+export function kaikkiEntry(d: any) {
+ const senses = (d.senses ?? []).filter((s: any) => s.glosses?.length);
+ const sounds = d.sounds ?? [];
+ return {
+ pos: d.pos as string,
+ glosses: senses.map((s: any) => s.glosses.join('; ')).slice(0, 6) as string[],
+ formOf: senses.length > 0 && senses.every((s: any) => s.form_of || s.tags?.includes('form-of')),
+ ipa: [...new Set(sounds.map((s: any) => s.ipa).filter(Boolean))].slice(0, 2) as string[],
+ audio: sounds.map((s: any) => upload(s.mp3_url) ?? upload(s.ogg_url)).find(Boolean) ?? null,
+ tr: d.forms?.find((f: any) => f.tags?.includes('romanization'))?.form ?? null,
+ ety: d.etymology_text ? etymology(String(d.etymology_text)) : null,
+ synonyms: (d.synonyms ?? []).map((s: any) => s.word).filter(Boolean).slice(0, 8) as string[],
+ };
}
-// fa./ar.wiktionary: "# definition" lines outside etymology/translation sections; on ar.wiktionary only
-// the Arabic section, and undiacritised pages that just point to entries ("* [[عِشْق]]") give those links
+// etymology text without Wiktionary's "Etymology tree …" diagram summary; cut at 300 characters, never
+// leaving half a surrogate pair
+const etymology = (t: string) =>
+ (t.startsWith('Etymology tree') ? t.slice(Math.max(0, t.search(/\b(Borrowed|Inherited|From|Learned|Derived|Semi-learned|Calque|Compound)\b/))) : t)
+ .slice(0, 300).toWellFormed();
+
+// English glosses of an Urdu entry as pivot keys: "fate, destiny" -> fate, destiny (short, lowercase)
+export const glossKeys = (glosses: string[]) => [...new Set(glosses.flatMap((g) => g.replace(/\([^)]*\)/g, '').split(/[,;]/))
+ .map((s) => s.trim().toLowerCase().replace(/^(to|a|an|the) /, '')).filter((s) => /^[a-z][a-z' -]{1,30}$/.test(s) && s.split(' ').length <= 3))];
+
+// ur.wiktionary Urdu entries: numbered lines under ==معانی==, else "# " lines, else the opening prose;
+// origin from "(عربی)" on the first line
+export function urduEntry(wt: string) {
+ const m = wt.split(/==\s*معانی\s*==/)[1]?.split(/\n==/)[0] ?? '';
+ let defs = m.split('\n').map((l) => l.match(/^\s*(?:\d+[.)-]|#)\s*(.+)/)?.[1]).filter(Boolean).map((l) => unlink(l!));
+ if (!defs.length) defs = definitions(wt, 'ur').defs;
+ if (!defs.length) {
+ const prose = wt.split('\n').find((l) => l.trim().startsWith("'''") && l.length > 20);
+ if (prose) defs = [unlink(stripTemplates(prose)).slice(0, 300)];
+ }
+ const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null;
+ const english = wt.match(/^\s*انگریزی\s*:\s*([A-Za-z].+)$/m)?.[1].trim() ?? null; // "== تراجم ==" pages
+ return { defs: defs.slice(0, 6), origin, links: [] as string[], ...(english && { english }) };
+}
+
+// ur.wiktionary English entries ("north": "# [[شمالی]]۔"): the Urdu words, for the pivot through English
+export const isEnglish = (title: string) => /^[A-Za-z][A-Za-z' -]*$/.test(title);
+export function englishToUrdu(wt: string) {
+ const lines = wt.split('\n').filter((l) => /^#(?![:*])/.test(l)).join(' ');
+ return [...new Set([...lines.matchAll(/\[\[(?:[^\]|]*\|)?([^\]]+)\]\]/g)].map((m) => m[1].trim())
+ .filter((w) => /^[\p{Script=Arabic}\s]+$/u.test(w)))].slice(0, 10);
+}
+const stripTemplates = (t: string) => {
+ while (/\{\{[^{}]*\}\}/.test(t)) t = t.replace(/\{\{[^{}]*\}\}/g, '');
+ return t;
+};
+
+// fa./ar.wiktionary: "# definition" lines outside etymology/translation sections; on ar.wiktionary only the
+// Arabic section, and undiacritised pages that just point to entries ("* [[عِشْق]]") give those links
export function definitions(wt: string, code: string) {
if (code === 'ar' && wt.includes('{{اللغة|')) wt = wt.split(/==\s*\{\{اللغة\|/).find((x) => x.startsWith('عربية')) ?? '';
const defs: string[] = [], links: string[] = [];
@@ -55,102 +96,214 @@ export function definitions(wt: string, code: string) {
if (only) { links.push(only[1]); continue; }
const m = line.match(/^#(?![:*])\s*(.+)/);
if (!m) continue;
- let t = m[1];
- while (/\{\{[^{}]*\}\}/.test(t)) t = t.replace(/\{\{[^{}]*\}\}/g, '');
- t = unlink(t).replace(/^[\s.،:-]+|[\s]+$/g, '');
+ const t = unlink(stripTemplates(m[1])).replace(/^[\s.،:-]+|\s+$/g, '');
if (t.length >= 3 && !/^-+$/.test(t)) defs.push(t);
}
- return { defs: defs.slice(0, 5), links };
+ return { defs: defs.slice(0, 6), origin: null as string | null, links: links.slice(0, 5) };
}
-// meanings from the language's own Wiktionary: the first spelling with an entry (following one pointer hop)
-async function native(code: string, word: string) {
- const host = `${code}.wiktionary.org`;
- for (const f of formsIn(code, word)) {
- const wt = await wikitext(host, f);
- if (!wt) continue;
- if (code === 'ur') {
- const e = urduEntry(wt);
- if (e.meanings.length) return { defs: e.meanings, origin: e.origin, url: `https://${host}/wiki/${encodeURIComponent(f)}` };
- continue;
+export const ownEntry = (code: string, wt: string) => (code === 'ur' ? urduEntry(wt) : definitions(wt, code));
+
+// pages of a MediaWiki XML dump (articles only, no redirects)
+export function* dumpPages(xml: string) {
+ for (const m of xml.matchAll(/([\s\S]*?)<\/page>/g)) {
+ const p = m[1];
+ if (!/0<\/ns>/.test(p) || /([^<]*)<\/title>/)?.[1];
+ const text = p.match(/]*>([\s\S]*?)<\/text>/)?.[1];
+ if (title && text) yield { title: xmlText(title), text: xmlText(text) };
+ }
+}
+const xmlText = (s: string) => s.replace(/</g, '<').replace(/>/g, '>').replace(/"/g, '"').replace(/'/g, "'").replace(/&/g, '&');
+
+// ---- import and sync ----
+
+const meta = async (k: string) => (await pool.query('SELECT value FROM dict_meta WHERE name = $1', [k])).rows[0]?.value ?? null;
+const setMeta = (k: string, v: string) =>
+ pool.query('INSERT INTO dict_meta (name, value) VALUES ($1, $2) ON CONFLICT (name) DO UPDATE SET value = $2', [k, v]);
+
+async function insertRows(client: any, rows: unknown[][]) {
+ for (let i = 0; i < rows.length; i += 500) {
+ const chunk = rows.slice(i, i + 500), values: unknown[] = [];
+ const sql = chunk.map((r, j) => { values.push(...r); const o = j * 5; return `($${o + 1},$${o + 2},$${o + 3},$${o + 4},$${o + 5})`; });
+ await client.query(`INSERT INTO wiktionary (lang, source, title, key, data) VALUES ${sql.join(',')}`, values);
+ }
+}
+
+// replace one source's rows in a transaction (readers see the old data until it commits)
+async function replace(lang: string, source: string, fill: (add: (title: string, data: object) => Promise, client: any) => Promise) {
+ const client = await pool.connect();
+ let n = 0;
+ try {
+ await client.query('BEGIN');
+ await client.query('DELETE FROM wiktionary WHERE lang = $1 AND source = $2', [lang, source]);
+ if (lang === 'ur') await client.query('DELETE FROM ur_glosses WHERE source = $1', [source]);
+ let batch: unknown[][] = [];
+ await fill(async (title, data) => {
+ batch.push([lang, source, title, key(title), data]); n++;
+ if (batch.length >= 2000) { await insertRows(client, batch); batch = []; }
+ }, client);
+ await insertRows(client, batch);
+ await client.query('COMMIT');
+ } catch (e) {
+ await client.query('ROLLBACK');
+ throw e;
+ } finally {
+ client.release();
+ }
+ return n;
+}
+
+async function lastModified(url: string) {
+ const res = await fetch(url, { method: 'HEAD', headers: { 'user-agent': UA } });
+ if (!res.ok) throw new Error(`${res.status} ${url}`);
+ return res.headers.get('last-modified') ?? '';
+}
+
+async function importKaikki(lang: string) {
+ const res = await fetch(KAIKKI(lang), { headers: { 'user-agent': UA } });
+ if (!res.ok || !res.body) throw new Error(`${res.status} ${KAIKKI(lang)}`);
+ return replace(lang, 'en', async (add, client) => {
+ const glossRows: string[][] = [];
+ for await (const line of createInterface({ input: Readable.fromWeb(res.body as any), crlfDelay: Infinity })) {
+ if (!line) continue;
+ const d = JSON.parse(line), e = kaikkiEntry(d);
+ if (!e.glosses.length && !e.ipa.length) continue;
+ await add(d.word, e);
+ if (lang === 'ur' && !e.formOf) for (const g of glossKeys(e.glosses.slice(0, 3))) glossRows.push([g, d.word]);
}
- let { defs, links } = definitions(wt, code), title = f;
- if (!defs.length && links[0]) ({ defs } = definitions(await wikitext(host, (title = links[0])), code));
- if (defs.length) return { defs, origin: null, url: `https://${host}/wiki/${encodeURIComponent(title)}` };
+ await insertGlosses(client, 'en', glossRows);
+ });
+}
+
+async function insertGlosses(client: any, source: string, rows: string[][]) {
+ for (let i = 0; i < rows.length; i += 1000) {
+ const chunk = rows.slice(i, i + 1000);
+ await client.query(`INSERT INTO ur_glosses (gloss, word, source) VALUES ${chunk.map((_, j) => `($${j * 2 + 1},$${j * 2 + 2},'${source}')`).join(',')}`, chunk.flat());
}
- return null;
}
-// ur.wiktionary: numbered lines under ==معانی==, origin from "(عربی)" on the first line
-export function urduEntry(wt: string) {
- const m = wt.split(/==\s*معانی\s*==/)[1]?.split(/\n==/)[0] ?? '';
- const meanings = m.split('\n').map((l) => l.match(/^\s*(?:\d+[.)-]|#)\s*(.+)/)?.[1]).filter(Boolean).map((l) => unlink(l!)).slice(0, 6);
- const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null;
- return { meanings, origin };
+async function importDump(lang: string) {
+ const bz = spawn('sh', ['-c', `curl -sfL -A '${UA}' '${DUMP(lang)}' | bzcat`]);
+ return replace(lang, 'own', async (add, client) => {
+ let buf = '';
+ const glossRows: string[][] = []; // ur.wiktionary English entries -> the pivot table
+ for await (const chunk of bz.stdout.setEncoding('utf8')) {
+ buf += chunk;
+ const end = buf.lastIndexOf('');
+ if (end < 0) continue;
+ for (const p of dumpPages(buf.slice(0, end + 7))) {
+ if (lang === 'ur' && isEnglish(p.title)) {
+ for (const w of englishToUrdu(p.text)) glossRows.push([p.title.toLowerCase(), w]);
+ continue;
+ }
+ const e = ownEntry(lang, p.text);
+ if (e.defs.length || e.links.length || 'english' in e) await add(p.title, e);
+ }
+ buf = buf.slice(end + 7);
+ }
+ const code: number = await new Promise((r) => (bz.exitCode !== null ? r(bz.exitCode) : bz.on('close', r)));
+ if (code !== 0) throw new Error(`dump download/decompress failed for ${lang} (exit ${code})`);
+ await insertGlosses(client, 'own', glossRows);
+ });
}
-// en.wiktionary page HTML: per language, IPA, headword transliteration, audio
-export function pronunciations(html: string) {
- const out: Record = {};
- const parts = html.split(/]*id="([^"]+)"/);
- for (let i = 1; i < parts.length; i += 2) {
- const code = Object.entries(LANGS).find(([, n]) => n === parts[i])?.[0];
- if (!code) continue;
- const body = parts[i + 1];
- const ipa = [...new Set([...body.matchAll(/class="IPA[^"]*"[^>]*>([^<]+)/g)].map((m) => m[1]).filter((x) => /^[/[]/.test(x)))].slice(0, 2);
- const tr = body.match(new RegExp(`lang="${code}-Latn" class="headword-tr[^"]*"[^>]*>([^<]+)`))?.[1] ?? null;
- const a = body.match(/"(?:https:)?(\/\/upload\.wikimedia\.org\/[^"]+\.(?:ogg|oga|mp3|wav))"/)?.[1];
- out[code] = { ipa, tr: tr && plain(tr), audio: a ? 'https:' + a : null };
+// changes on ur./fa./ar.wiktionary since the last run, re-read from the API (titles in batches of 50)
+async function recentChanges(lang: string) {
+ const api = `https://${lang}.wiktionary.org/w/api.php`;
+ const since = await meta(`rc:${lang}`), now = new Date().toISOString();
+ if (!since) return setMeta(`rc:${lang}`, now).then(() => 0); // first run: the dump is the baseline
+ const titles = new Set();
+ let cont: Record = {};
+ do {
+ const q = new URLSearchParams({ action: 'query', list: 'recentchanges', rcnamespace: '0', rctype: 'edit|new', rcprop: 'title',
+ rclimit: '500', rcdir: 'newer', rcstart: since, rcend: now, format: 'json', formatversion: '2', ...cont });
+ const d: any = await (await fetch(`${api}?${q}`, { headers: { 'user-agent': UA } })).json();
+ for (const c of d.query?.recentchanges ?? []) titles.add(c.title);
+ cont = d.continue ?? {};
+ } while (cont.rccontinue);
+ const list = [...titles];
+ for (let i = 0; i < list.length; i += 50) {
+ const q = new URLSearchParams({ action: 'query', prop: 'revisions', rvprop: 'content', rvslots: 'main',
+ titles: list.slice(i, i + 50).join('|'), format: 'json', formatversion: '2' });
+ const d: any = await (await fetch(api, { method: 'POST', body: q, headers: { 'user-agent': UA } })).json();
+ for (const p of d.query?.pages ?? []) {
+ const wt = p.revisions?.[0]?.slots?.main?.content;
+ if (lang === 'ur' && isEnglish(p.title)) {
+ const g = p.title.toLowerCase();
+ await pool.query(`DELETE FROM ur_glosses WHERE source = 'own' AND gloss = $1`, [g]);
+ for (const w of wt ? englishToUrdu(wt) : []) await pool.query(`INSERT INTO ur_glosses (gloss, word, source) VALUES ($1, $2, 'own')`, [g, w]);
+ continue;
+ }
+ await pool.query(`DELETE FROM wiktionary WHERE lang = $1 AND source = 'own' AND title = $2`, [lang, p.title]);
+ const e = wt && !/^#(REDIRECT|تحويل|تغییر)/i.test(wt) ? ownEntry(lang, wt) : null;
+ if (e && (e.defs.length || e.links.length || 'english' in e))
+ await pool.query(`INSERT INTO wiktionary (lang, source, title, key, data) VALUES ($1, 'own', $2, $3, $4)`, [lang, p.title, key(p.title), e]);
+ }
}
- return out;
+ await setMeta(`rc:${lang}`, now);
+ return list.length;
}
-// Urdu equivalents of single-word English glosses, from their translation tables
-async function pivot(word: string, glosses: string[]) {
- const seen = new Set([word.replace(DIAC, '')]); // the word itself, and duplicates differing only in diacritics
- const words = [...new Set(glosses.flatMap((g) => g.split(/[,;]/)).map((s) => s.trim()).filter((s) => /^[a-z]+$/.test(s)))].slice(0, 3); // lowercase: no proper nouns
- const found = await Promise.all(words.map(async (gloss) => {
- const wt = await wikitext('en.wiktionary.org', gloss);
- const lines = [...wt.matchAll(/^\*:? Urdu: (.*)$/gm)].map((m) => m[1]).join(' ');
- const urdu = [...lines.matchAll(/\{\{t\+?\|ur\|([^|}]+)/g)].map((m) => m[1].trim())
- .filter((u) => !seen.has(u.replace(DIAC, '')) && seen.add(u.replace(DIAC, ''))).slice(0, 8);
- return { gloss, urdu };
- }));
- return found.filter((f) => f.urdu.length);
+// full re-import of each source whose upstream file changed, then the daily recent changes
+export async function sync(log = console.log) {
+ for (const lang of LANGS) {
+ for (const [source, url, run] of [['en', KAIKKI(lang), importKaikki], ['own', DUMP(lang), importDump]] as const) {
+ const lm = await lastModified(url), k = `file:${lang}:${source}`;
+ if (lm && lm === (await meta(k))) continue;
+ const t = Date.now(), n = await run(lang);
+ await setMeta(k, lm);
+ if (source === 'own') await setMeta(`rc:${lang}`, new Date(lm).toISOString()); // changes after the dump
+ log(`${lang} ${source}: ${n} entries (${Math.round((Date.now() - t) / 1000)} s)`);
+ }
+ log(`${lang} recent changes: ${await recentChanges(lang)} pages`);
+ }
}
-async function fetchWord(word: string) {
- let title: string | null = null, defs: any = null;
- for (const f of forms(word)) if ((defs = await get(rest('definition', f)))) { title = f; break; }
- const [pron, ...own] = await Promise.all([
- title ? get(rest('html', title), false).then((h) => pronunciations(h ?? '')) : {},
- ...Object.keys(LANGS).map((c) => native(c, word)),
- ]) as [Record, ...(Awaited>)[]];
- const langs = Object.keys(LANGS).map((code, i) => ({
- code, ...(pron[code] ?? { ipa: [], tr: null, audio: null }),
- meanings: own[i]?.defs ?? [], origin: own[i]?.origin ?? null, source: own[i]?.url ?? null,
- senses: (defs?.[code] ?? []).map((e: any) => ({
- pos: e.partOfSpeech, defs: e.definitions.map((d: any) => plain(d.definition)).filter(Boolean).slice(0, 4),
- })).filter((s: any) => s.defs.length).slice(0, 3),
- })).filter((l) => l.meanings.length || l.senses.length || l.ipa.length || l.tr);
- const glosses = langs.find((l) => l.senses.length)?.senses[0].defs.slice(0, 2) ?? []; // the primary meaning only
- const hasUrdu = langs.some((l) => l.code === 'ur' && l.meanings.length);
- return {
- word, langs,
- equivalents: hasUrdu ? [] : await pivot(word, glosses),
- sources: { en: title && `https://en.wiktionary.org/wiki/${encodeURIComponent(title)}` },
- found: langs.length > 0,
- };
-}
+// ---- lookup ----
+
+const page = (lang: string, source: string, title: string) =>
+ source === 'en' ? `https://en.wiktionary.org/wiki/${encodeURIComponent(title)}#${NAME[lang as 'ur']}` : `https://${lang}.wiktionary.org/wiki/${encodeURIComponent(title)}`;
export async function lookup(word: string) {
- const hit = await pool.query(`SELECT data FROM dictionary WHERE word = $1 AND fetched_at > now() - interval '30 days'`, [word]);
- if (hit.rows[0]) return hit.rows[0].data;
- const data = await fetchWord(word);
- if (!data.found) return data; // not found (or Wiktionary unreachable): don't cache
- await pool.query(
- `INSERT INTO dictionary (word, data) VALUES ($1, $2) ON CONFLICT (word) DO UPDATE SET data = $2, fetched_at = now()`,
- [word, data],
- );
- return data;
+ const k = key(word);
+ const { rows } = await pool.query('SELECT lang, source, title, data FROM wiktionary WHERE key = $1 ORDER BY id', [k]);
+ // ar.wiktionary pointer pages: follow to the diacritised entries
+ const pointers = rows.filter((r) => r.source === 'own' && !r.data.defs.length).flatMap((r) => r.data.links.map((l: string) => [r.lang, l]));
+ if (pointers.length) {
+ const more = await pool.query(
+ `SELECT lang, source, title, data FROM wiktionary WHERE source = 'own' AND (lang, title) IN (SELECT * FROM unnest($1::text[], $2::text[]))`,
+ [pointers.map((p) => p[0]), pointers.map((p) => p[1])]);
+ rows.push(...more.rows);
+ }
+ const langs = LANGS.map((code) => {
+ const en = rows.filter((r) => r.lang === code && r.source === 'en');
+ const own = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.defs.length);
+ const lemmas = en.filter((r) => !r.data.formOf), use = lemmas.length ? lemmas : en;
+ const first = (f: string) => use.map((r) => r.data[f]).find((v) => (Array.isArray(v) ? v.length : v)) ?? null;
+ // ur.wiktionary's English translation lines ("انگریزی : …")
+ const translated = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.english).map((r) => r.data.english);
+ return {
+ code, ipa: first('ipa') ?? [], tr: first('tr'), audio: first('audio'), ety: first('ety'),
+ synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))].slice(0, 8),
+ meanings: own.flatMap((r) => r.data.defs).slice(0, 6), origin: own.find((r) => r.data.origin)?.data.origin ?? null,
+ source: own[0] ? page(code, 'own', own[0].title) : null,
+ senses: [...use.map((r) => ({ pos: r.data.pos, defs: r.data.glosses.slice(0, 4) })).filter((s) => s.defs.length).slice(0, 3),
+ ...(translated.length ? [{ pos: 'translation', defs: translated }] : [])],
+ en: en[0] ? page(code, 'en', en[0].title) : null,
+ };
+ }).filter((l) => l.meanings.length || l.senses.length || l.ipa.length || l.tr);
+ // no Urdu meaning: Urdu words sharing the primary English meaning
+ let equivalents: { gloss: string; urdu: string[] }[] = [];
+ if (!langs.some((l) => l.code === 'ur' && l.meanings.length)) {
+ const glosses = glossKeys(langs.find((l) => l.senses.length)?.senses[0].defs.slice(0, 2) ?? []).slice(0, 4);
+ if (glosses.length) {
+ const { rows: eq } = await pool.query('SELECT gloss, word FROM ur_glosses WHERE gloss = ANY($1)', [glosses]);
+ const seen = new Set([k]);
+ equivalents = glosses.map((g) => ({
+ gloss: g, urdu: eq.filter((r) => r.gloss === g && !seen.has(key(r.word)) && seen.add(key(r.word))).map((r) => r.word).slice(0, 8),
+ })).filter((e) => e.urdu.length);
+ }
+ }
+ return { word, langs, equivalents, found: langs.length > 0 };
}
diff --git a/db/schema.sql b/db/schema.sql
index dc20eff7..b9edef02 100644
--- a/db/schema.sql
+++ b/db/schema.sql
@@ -58,9 +58,25 @@ CREATE TABLE IF NOT EXISTS verses (
PRIMARY KEY (poem_id, vorder)
);
--- Wiktionary lookups for the reading sidebar (api/src/dictionary.ts), refreshed after 30 days
-CREATE TABLE IF NOT EXISTS dictionary (
- word text PRIMARY KEY,
- data jsonb NOT NULL,
- fetched_at timestamptz NOT NULL DEFAULT now()
+-- Wiktionary for the word sidebar (api/src/dictionary.ts; filled and kept current by dict-sync.ts)
+DROP TABLE IF EXISTS dictionary; -- the earlier live-lookup cache
+CREATE TABLE IF NOT EXISTS wiktionary (
+ id bigserial PRIMARY KEY,
+ lang text NOT NULL, -- the word's language: ur | fa | ar
+ source text NOT NULL, -- en (en.wiktionary, via kaikki.org) | own (that language's Wiktionary)
+ title text NOT NULL, -- headword / page title
+ key text NOT NULL, -- spelling-insensitive lookup key (dictionary.ts key())
+ data jsonb NOT NULL
+);
+CREATE INDEX IF NOT EXISTS wiktionary_key ON wiktionary(key);
+CREATE INDEX IF NOT EXISTS wiktionary_title ON wiktionary(lang, source, title);
+CREATE TABLE IF NOT EXISTS ur_glosses ( -- English gloss -> Urdu word, for the pivot through English
+ gloss text NOT NULL,
+ word text NOT NULL
+);
+ALTER TABLE ur_glosses ADD COLUMN IF NOT EXISTS source text NOT NULL DEFAULT 'en'; -- en | own (ur.wiktionary English entries)
+CREATE INDEX IF NOT EXISTS ur_glosses_gloss ON ur_glosses(gloss);
+CREATE TABLE IF NOT EXISTS dict_meta ( -- upstream file versions and recent-changes timestamps
+ name text PRIMARY KEY,
+ value text NOT NULL
);
diff --git a/deploy/sync.sh b/deploy/sync.sh
index 8ede1c40..bfc01e8b 100755
--- a/deploy/sync.sh
+++ b/deploy/sync.sh
@@ -4,6 +4,7 @@
# 1. update the divan-data checkout and fetch new/edited works from Wikisource (incremental)
# 2. rebuild its search index and site export
# 3. load the export into PostgreSQL (upserts; the site reads it live)
+# 3b. sync the word dictionary from Wiktionary (api/src/dict-sync.ts; first run imports ~700 MB)
# 4. optionally commit + push the refreshed data (DIVAN_DATA_PUSH=1, needs git push access)
#
# Schedule with cron, e.g. daily at 03:15:
@@ -11,7 +12,7 @@
#
# Settings (env): DIVAN_DATA_DIR (default /opt/divan-data), DIVAN_APP_DIR (default: this repo),
# DATABASE_URL (Postgres, as for the API), DIVAN_DATA_PUSH=1 to push data changes.
-# Needs: git, python3 (stdlib only), node 24+.
+# Needs: git, python3 (stdlib only), node 24+, curl and bzcat (dictionary dumps).
set -euo pipefail
APP_DIR=${DIVAN_APP_DIR:-$(cd "$(dirname "$0")/.." && pwd)}
@@ -36,4 +37,8 @@ fi
cd "$APP_DIR/api"
[ -d node_modules ] || npm ci --omit=dev --silent
node src/import.ts "$DATA_DIR"
+
+# word dictionary (Wiktionary): re-imports sources that changed upstream, then the day's edits.
+# A failure here must not fail the content sync.
+node src/dict-sync.ts || echo "$(date -u +%FT%TZ) dictionary sync failed"
echo "== $(date -u +%FT%TZ) done"
diff --git a/web/src/layouts/Base.astro b/web/src/layouts/Base.astro
index 431b5999..45b696ee 100644
--- a/web/src/layouts/Base.astro
+++ b/web/src/layouts/Base.astro
@@ -120,7 +120,8 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
const dict = document.getElementById('dict'), dictBody = dict.querySelector('.dict-body'), main = document.querySelector('main');
const LANG = { ur: 'اردو', fa: 'فارسی', ar: 'عربی' };
const POS = { Noun: 'اسم', 'Proper noun': 'اسم معرفہ', Verb: 'فعل', Adjective: 'صفت', Adverb: 'متعلق فعل', Pronoun: 'ضمیر',
- Postposition: 'حرف جار', Preposition: 'حرف جار', Conjunction: 'حرف عطف', Interjection: 'حرف ندا', Particle: 'حرف' };
+ Postposition: 'حرف جار', Preposition: 'حرف جار', Conjunction: 'حرف عطف', Interjection: 'حرف ندا', Particle: 'حرف',
+ Translation: 'ترجمہ (اردو ویکی لغت)', Phrase: 'فقرہ', Proverb: 'کہاوت', Name: 'اسم معرفہ' };
const el = (tag, cls, text, attrs) => {
const e = document.createElement(tag);
if (cls) e.className = cls;
@@ -134,7 +135,8 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
current = w; dict.hidden = false; document.body.classList.add('dict-open');
dictBody.replaceChildren(el('h2', null, w), el('p', 'muted', 'تلاش جاری ہے…'));
let d = null;
- try { const r = await fetch('/api/word?w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {}
+ // v: bump when the response format changes (responses are browser-cached for a day)
+ try { const r = await fetch('/api/word?v=2&w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {}
if (w !== current) return; // a newer selection won
const out = [el('h2', null, w)];
if (!d || !d.found) {
@@ -171,14 +173,20 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
if (l.source) label.append(el('a', null, 'ماخذ', { href: l.source, target: '_blank', rel: 'noopener' }));
sec.append(label, list(l.meanings, null, { lang: l.code }));
}
- for (const s of l.senses) sec.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' }));
+ for (const s of l.senses) sec.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos.charAt(0).toUpperCase() + s.pos.slice(1)] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' }));
+ if (l.synonyms.length) sec.append(el('p', 'pos', 'مترادفات'), el('p', null, l.synonyms.join('، ')));
+ if (l.ety) sec.append(el('p', 'pos', 'اشتقاق'), el('p', 'en ety', l.ety, { dir: 'ltr', lang: 'en' }));
+ if (l.en) {
+ const p = el('p', 'pos');
+ p.append(el('a', null, 'انگریزی ویکی لغت', { href: l.en, target: '_blank', rel: 'noopener' }));
+ sec.append(p);
+ }
out.push(sec);
if (l.code === 'ur' && d.equivalents.length) out.push(equivalents());
}
if (d.equivalents.length && !d.langs.some((l) => l.code === 'ur')) out.splice(1, 0, equivalents());
const src = el('p', 'src muted');
- src.append('ویکی لغت · ');
- if (d.sources.en) src.append(el('a', null, 'انگریزی ویکی لغت', { href: d.sources.en, target: '_blank', rel: 'noopener' }), ' · ');
+ src.append('ویکی لغت (Wiktionary) · ');
src.append('CC BY-SA');
dictBody.replaceChildren(...out, src);
}
diff --git a/web/src/styles/global.css b/web/src/styles/global.css
index ab19cca3..7e3e768d 100644
--- a/web/src/styles/global.css
+++ b/web/src/styles/global.css
@@ -146,6 +146,7 @@ h1 + .muted { text-align: center; margin-top: 0; }
.dict .pos { font-size: .8rem; color: var(--muted); margin: 8px 0 0; }
.dict .note { font-size: .8rem; margin: 0; }
.dict .src { font-size: .78rem; margin-top: 18px; }
+.dict .ety { text-align: left; color: var(--muted); font-size: .8rem; }
.dict-close { position: sticky; top: 0; float: left; border: 0; background: none; color: var(--muted); font-size: 1rem; cursor: pointer; }
.dict .play { border: 1px solid var(--border); background: var(--inner); border-radius: 8px; cursor: pointer; padding: 0 6px; }
/* wide screens: the page moves over instead of being covered */