From 7fcc0ee7fa6fe2ca65a8fda9fbc55dda3f36ca3a Mon Sep 17 00:00:00 2001 From: Anas Rashid Date: Thu, 8 Oct 2026 21:48:56 +0200 Subject: [PATCH] Word dictionary: full Wiktionary data for Urdu, Persian and Arabic in PostgreSQL, kept in sync - en.wiktionary entries via kaikki.org (meanings, IPA, transliteration, audio, etymology, synonyms) - ur./fa./ar.wiktionary dumps for meanings in each language (incl. more Urdu entry formats) - English->Urdu pivot table from Urdu entries' glosses and ur.wiktionary's English entries - dict-sync.ts: re-imports sources when upstream publishes new files, applies daily recent changes; run from deploy/sync.sh - lookups now read the database (2-7 ms) instead of calling Wikimedia live Co-Authored-By: Claude Opus 5.5 --- README.md | 4 +- api/package.json | 3 +- api/src/dict-sync.ts | 10 + api/src/dictionary.test.ts | 75 +++++-- api/src/dictionary.ts | 389 ++++++++++++++++++++++++++----------- db/schema.sql | 26 ++- deploy/sync.sh | 7 +- web/src/layouts/Base.astro | 18 +- web/src/styles/global.css | 1 + 9 files changed, 381 insertions(+), 152 deletions(-) create mode 100644 api/src/dict-sync.ts diff --git a/README.md b/README.md index 9208d68a..83a5a022 100644 --- a/README.md +++ b/README.md @@ -42,6 +42,8 @@ The import upserts, so re-running it after a divan-data sync applies the changes 15 3 * * * DATABASE_URL=postgres://divan:...@localhost:5432/divan /opt/divan/deploy/sync.sh >> /var/log/divan-sync.log 2>&1 ``` +The same script keeps the word dictionary current (`npm run dict-sync` in `api/`): the full Wiktionary data for Urdu, Persian and Arabic (English Wiktionary via [kaikki.org](https://kaikki.org), and the Urdu, Persian and Arabic Wiktionary dumps) is re-imported when upstream publishes new files, and each day's Wiktionary edits are applied from recent changes. The first run downloads about 700 MB. + Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DIVAN_APP_DIR` (default: this repo), `DIVAN_DATA_PUSH=1` to also commit and push data changes (needs git push access). Needs git, python3 and Node 24+. ## API @@ -51,7 +53,7 @@ Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DI | `GET /api/poets` | all poets | | `GET /api/page?url=/p238/...` | the poet, category or poem at a site URL (breadcrumbs, children, verses, prev/next) | | `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised; exact phrase first; each with the best-matching couplet or paragraph (`snippet`) | -| `GET /api/word?w=` | one word's meanings and pronunciation from Wiktionary (Urdu, Persian, Arabic in that order; English meanings; Urdu equivalents via English when Urdu Wiktionary has none), cached 30 days | +| `GET /api/word?w=` | one word's meanings and pronunciation from the local Wiktionary data (Urdu, Persian, Arabic in that order; English meanings; Urdu equivalents via English when Urdu Wiktionary has none) | | `GET /health` | database check | Search normalises both stored text and queries: Arabic ي/ك/ه → Urdu ی/ک/ہ, ۂ/ۓ, diacritics and the Urdu full stop removed; do-chashmi ھ stays distinct. diff --git a/api/package.json b/api/package.json index 4c7a48b7..c5206c0f 100644 --- a/api/package.json +++ b/api/package.json @@ -9,7 +9,8 @@ "scripts": { "start": "node src/server.ts", "import": "node src/import.ts", - "test": "node --test src/*.test.ts" + "test": "node --test src/*.test.ts", + "dict-sync": "node src/dict-sync.ts" }, "dependencies": { "fastify": "^5.12.5", diff --git a/api/src/dict-sync.ts b/api/src/dict-sync.ts new file mode 100644 index 00000000..494fada0 --- /dev/null +++ b/api/src/dict-sync.ts @@ -0,0 +1,10 @@ +// Import / sync the Wiktionary data (see dictionary.ts). First run imports everything (about 1 GB of downloads); +// later runs re-import only sources that changed upstream and apply the day's Wiktionary edits. +// node src/dict-sync.ts +import { readFile } from 'node:fs/promises'; +import { pool } from './db.ts'; +import { sync } from './dictionary.ts'; + +await pool.query(await readFile(new URL('../../db/schema.sql', import.meta.url), 'utf8')); +await sync((line) => console.log(`${new Date().toISOString()} ${line}`)); +await pool.end(); diff --git a/api/src/dictionary.test.ts b/api/src/dictionary.test.ts index 4036aa44..6526eda8 100644 --- a/api/src/dictionary.test.ts +++ b/api/src/dictionary.test.ts @@ -1,34 +1,67 @@ import { test } from 'node:test'; import assert from 'node:assert/strict'; -import { forms, urduEntry, pronunciations } from './dictionary.ts'; +import { key, kaikkiEntry, glossKeys, urduEntry, definitions, dumpPages } from './dictionary.ts'; -test('spelling forms: diacritics dropped, final noon ghunna tried as noon', () => { - assert.deepEqual(forms('ناداں'), ['ناداں', 'نادان']); - assert.deepEqual(forms('وِصال'), ['وِصال', 'وصال']); +test('lookup key: spelling variants across Urdu, Persian and Arabic meet', () => { + const same = (a: string, b: string) => assert.equal(key(a), key(b), `${a} vs ${b}`); + same('ناداں', 'نادان'); + same('قِسْمَت', 'قسمت'); + same('نگاہ', 'نگاه'); + same('معنی', 'معنى'); + same('كتاب', 'کتاب'); + assert.notEqual(key('کھ'), key('کہ'), 'do-chashmi heh stays distinct'); + assert.notEqual(key('دیوانے'), key('دیوانی'), 'bari ye stays distinct'); +}); + +test('kaikki entry: glosses, IPA, audio, romanisation, form-of', () => { + const e = kaikkiEntry({ + word: 'قسمت', pos: 'noun', etymology_text: 'From Arabic', + senses: [{ glosses: ['fate, destiny'] }, { glosses: ['division'] }, { links: [] }], + sounds: [{ ipa: '/qɪs.mət̪/' }, { audio: 'x.wav', mp3_url: 'https://upload.wikimedia.org/x.mp3' }, { mp3_url: 'https://evil.example/x.mp3' }], + forms: [{ form: 'قِسْمَت', tags: ['canonical'] }, { form: 'qismat', tags: ['romanization'] }], + }); + assert.deepEqual(e, { pos: 'noun', glosses: ['fate, destiny', 'division'], formOf: false, ipa: ['/qɪs.mət̪/'], + audio: 'https://upload.wikimedia.org/x.mp3', tr: 'qismat', ety: 'From Arabic', synonyms: [] }); + assert.equal(kaikkiEntry({ word: 'x', pos: 'noun', senses: [{ glosses: ['plural of y'], tags: ['form-of'] }] }).formOf, true); +}); + +test('pivot keys from English glosses', () => { + assert.deepEqual(glossKeys(['fate, destiny', 'to love (someone)', 'a very long description of something that is not a key']), ['fate', 'destiny', 'love']); }); test('ur.wiktionary entry: meanings and origin', () => { - const wt = "وِصال {وِصال} ([[عربی]])\n\nاسم [[نکرہ]]\n\n==معانی==\n\n1. بھیٹ، ملاقات، [[عاشق و معشوق]] کی محبت۔\n\n==مرکبات==\n\nوِصال کا دِن"; - assert.deepEqual(urduEntry(wt), { meanings: ['بھیٹ، ملاقات، عاشق و معشوق کی محبت۔'], origin: 'عربی' }); + const wt = 'وِصال {وِصال} ([[عربی]])\n\nاسم [[نکرہ]]\n\n==معانی==\n\n1. بھیٹ، ملاقات، [[عاشق و معشوق]] کی محبت۔\n\n==مرکبات==\n\nوِصال کا دِن'; + assert.deepEqual(urduEntry(wt), { defs: ['بھیٹ، ملاقات، عاشق و معشوق کی محبت۔'], origin: 'عربی', links: [] }); }); -test('en.wiktionary HTML: IPA, transliteration and audio per language', () => { - const html = '

Persian

/qis.ˈmat/-at' + - 'qismat' + - '

Urdu

/qɪs.mət̪/qismat'; - assert.deepEqual(pronunciations(html), { - fa: { ipa: ['/qis.ˈmat/'], tr: 'qismat', audio: 'https://upload.wikimedia.org/a/b.ogg' }, - ur: { ipa: ['/qɪs.mət̪/'], tr: 'qismat', audio: null }, - }); -}); - -test('fa./ar.wiktionary definitions: skip etymology, follow Arabic pointers', async () => { - const { definitions, formsIn } = await import('./dictionary.ts'); +test('fa./ar.wiktionary definitions: skip etymology, Arabic pointers', () => { const fa = '==فارسی==\n===ریشه‌شناسی===\n* [[عربی]]\n# قسمة\n===اسم===\n#بهره، نصیب.\n#:example\n#سرنوشت، تقدیر.\n====برگردان‌ها====\n# x y z'; assert.deepEqual(definitions(fa, 'fa').defs, ['بهره، نصیب.', 'سرنوشت، تقدیر.']); const ar = '== {{اللغة|عربية}} ==\nهل تقصد:\n* [[عِشْق]]\n* [[عَشَقَ]]\n----\n== {{اللغة|أردية}} ==\n# [[x]]'; - assert.deepEqual(definitions(ar, 'ar'), { defs: [], links: ['عِشْق', 'عَشَقَ'] }); + assert.deepEqual(definitions(ar, 'ar'), { defs: [], origin: null, links: ['عِشْق', 'عَشَقَ'] }); assert.deepEqual(definitions('# {{مصدر|عَشِقَ}}.\n# فرط الحب.', 'ar').defs, ['فرط الحب.']); - assert.deepEqual(formsIn('fa', 'نگاہ'), ['نگاه']); - assert.deepEqual(formsIn('ar', 'معنی'), ['معني', 'معنى']); +}); + +test('dump pages: articles only, entities decoded, redirects skipped', () => { + const xml = 'عشق0a & b <x>' + + 'Talk:x1tr0#R'; + assert.deepEqual([...dumpPages(xml)], [{ title: 'عشق', text: 'a & b ' }]); +}); + +test('etymology cut never leaves half a surrogate pair', () => { + const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'a'.repeat(299) + '𑀭𑀭' }).ety!; + assert.ok(ety.isWellFormed()); +}); + +test('ur.wiktionary: English entries give Urdu words; Urdu entries fall back to # lines or prose', async () => { + const { englishToUrdu, isEnglish } = await import('./dictionary.ts'); + assert.ok(isEnglish('north') && !isEnglish('شمال')); + assert.deepEqual(englishToUrdu('==انگریزی==\n===صفت===\n{{en-adjective}}\n# [[شمالی]]۔\n# [[شمال]] کی [[جانب]]۔\n#: [[x]]'), ['شمالی', 'شمال', 'جانب']); + assert.deepEqual(urduEntry('===اسم===\n# [[بہار]] کا [[موسم]]۔').defs, ['بہار کا موسم۔']); + assert.deepEqual(urduEntry("'''پھوڑی''' اس چادر کو کہتے ہیں جس پر بیٹھتے ہیں۔").defs, ['پھوڑی اس چادر کو کہتے ہیں جس پر بیٹھتے ہیں۔']); +}); + +test('etymology drops the "Etymology tree" summary', () => { + const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'Etymology tree Arabic قَسَمَ (qasama)bor. Urdu قِسْمَت Borrowed from Classical Persian قِسْمَت (qismat).' }).ety; + assert.equal(ety, 'Borrowed from Classical Persian قِسْمَت (qismat).'); }); diff --git a/api/src/dictionary.ts b/api/src/dictionary.ts index e38240f1..2336dae5 100644 --- a/api/src/dictionary.ts +++ b/api/src/dictionary.ts @@ -1,48 +1,89 @@ -// Word lookup for the reading sidebar, from Wiktionary (CC BY-SA). For Urdu, Persian and Arabic, in that order: -// - meanings in the language itself from its own Wiktionary (ur., fa., ar.wiktionary) -// - English meanings and pronunciation (IPA, transliteration, audio) from en.wiktionary -// With no Urdu meanings, Urdu equivalents are pivoted through English (the English glosses' translation tables). -// Found words are cached in PostgreSQL (dictionary table) for 30 days. +// Word dictionary for the reading sidebar: the full Wiktionary data for Urdu, Persian and Arabic, in PostgreSQL. +// source 'en': en.wiktionary entries (English meanings, IPA, transliteration, audio, etymology, synonyms), +// from the kaikki.org Wiktextract extracts (weekly) +// source 'own': each language's own Wiktionary (ur., fa., ar.wiktionary): meanings in that language, +// from the Wikimedia dumps (twice a month) plus recent changes (daily) +// Urdu equivalents for words with no Urdu meaning are pivoted through English: ur_glosses maps English words to +// Urdu words, from the English glosses of en.wiktionary's Urdu entries and ur.wiktionary's English entries. Data: CC BY-SA, Wiktionary contributors. +import { spawn } from 'node:child_process'; +import { createInterface } from 'node:readline'; +import { Readable } from 'node:stream'; import { pool } from './db.ts'; +export const LANGS = ['ur', 'fa', 'ar'] as const; +const NAME = { ur: 'Urdu', fa: 'Persian', ar: 'Arabic' } as const; const UA = 'Divan/2 (https://github.com/anas-rashid/divan)'; -const LANGS = { ur: 'Urdu', fa: 'Persian', ar: 'Arabic' } as const; -const DIAC = /[ً-ْٰٔٗ٘]/g; -const plain = (html: string) => - html.replace(/<[^>]+>/g, '').replace(/ /g, ' ').replace(/&/g, '&').replace(/&#\d+;/g, '').replace(/\s+/g, ' ').trim(); +const KAIKKI = (l: string) => `https://kaikki.org/dictionary/${NAME[l as 'ur']}/kaikki.org-dictionary-${NAME[l as 'ur']}.jsonl`; +const DUMP = (l: string) => `https://dumps.wikimedia.org/${l}wiktionary/latest/${l}wiktionary-latest-pages-articles.xml.bz2`; +const MARKS = /[ؐ-ًؚ-ٰٟۖ-ۭـ‌‍‎‏]/g; + +// spelling-insensitive key shared by Urdu, Persian and Arabic: no diacritics/tatweel, one heh, one yeh, one kaf, +// plain alef, noon ghunna as noon (ناداں = نادان, قسمت = قِسْمَت, نگاہ = نگاه, معنی = معنى). ے and ھ stay distinct. +export const key = (w: string) => + w.normalize('NFC').replace(MARKS, '') + .replace(/[ہۂۃةه]/g, 'ه').replace(/[يىی]/g, 'ی') + .replace(/ك/g, 'ک').replace(/[أإٱ]/g, 'ا').replace(/ں/g, 'ن').trim(); + const unlink = (wt: string) => wt.replace(/\[\[(?:[^\]|]*\|)?([^\]]*)\]\]/g, '$1').replace(/'''?/g, '').trim(); +const upload = (u?: string) => (u && u.startsWith('https://upload.wikimedia.org/') ? u : null); -async function get(url: string, json = true): Promise { - const res = await fetch(url, { headers: { 'user-agent': UA }, signal: AbortSignal.timeout(8000) }).catch(() => null); - if (!res?.ok) return null; - return json ? res.json() : res.text(); -} -const rest = (path: string, title: string) => `https://en.wiktionary.org/api/rest_v1/page/${path}/${encodeURIComponent(title)}`; -const wikitext = async (host: string, title: string): Promise => { - const d = await get(`https://${host}/w/api.php?action=parse&page=${encodeURIComponent(title)}&prop=wikitext&format=json&formatversion=2&redirects=1`); - return d?.parse?.wikitext ?? ''; -}; +// ---- parsing ---- -// spelling forms to try: as written, without diacritics, final noon ghunna as noon (ناداں -> نادان) -export const forms = (w: string) => { - const bare = w.replace(DIAC, '').replace(/ۂ/g, 'ہ'); - return [...new Set([w, bare, bare.replace(/ں$/, 'ن')])]; -}; - -// the word in Persian / Arabic spelling (ہ -> ه, ی -> ي, ک -> ك …); Arabic also tries final alef maqsura -const SPELL: Record = { - fa: [[/[\u06C1\u06C2\u06BE]/g, '\u0647'], [/[\u06D2\u064A]/g, '\u06CC'], [/\u0643/g, '\u06A9']], - ar: [[/[\u06C1\u06BE]/g, '\u0647'], [/\u06C2/g, '\u0629'], [/[\u06CC\u06D2]/g, '\u064A'], [/\u06A9/g, '\u0643']], -}; -export function formsIn(code: string, w: string) { - const base = forms(w); - if (!SPELL[code]) return base; - const conv = base.map((f) => SPELL[code].reduce((s, [re, to]) => s.replace(re, to), f)); - return [...new Set([...conv, ...(code === 'ar' ? conv.map((f) => f.replace(/\u064A$/, '\u0649')) : [])])]; +// a kaikki.org (Wiktextract) line -> the fields the sidebar shows +export function kaikkiEntry(d: any) { + const senses = (d.senses ?? []).filter((s: any) => s.glosses?.length); + const sounds = d.sounds ?? []; + return { + pos: d.pos as string, + glosses: senses.map((s: any) => s.glosses.join('; ')).slice(0, 6) as string[], + formOf: senses.length > 0 && senses.every((s: any) => s.form_of || s.tags?.includes('form-of')), + ipa: [...new Set(sounds.map((s: any) => s.ipa).filter(Boolean))].slice(0, 2) as string[], + audio: sounds.map((s: any) => upload(s.mp3_url) ?? upload(s.ogg_url)).find(Boolean) ?? null, + tr: d.forms?.find((f: any) => f.tags?.includes('romanization'))?.form ?? null, + ety: d.etymology_text ? etymology(String(d.etymology_text)) : null, + synonyms: (d.synonyms ?? []).map((s: any) => s.word).filter(Boolean).slice(0, 8) as string[], + }; } -// fa./ar.wiktionary: "# definition" lines outside etymology/translation sections; on ar.wiktionary only -// the Arabic section, and undiacritised pages that just point to entries ("* [[عِشْق]]") give those links +// etymology text without Wiktionary's "Etymology tree …" diagram summary; cut at 300 characters, never +// leaving half a surrogate pair +const etymology = (t: string) => + (t.startsWith('Etymology tree') ? t.slice(Math.max(0, t.search(/\b(Borrowed|Inherited|From|Learned|Derived|Semi-learned|Calque|Compound)\b/))) : t) + .slice(0, 300).toWellFormed(); + +// English glosses of an Urdu entry as pivot keys: "fate, destiny" -> fate, destiny (short, lowercase) +export const glossKeys = (glosses: string[]) => [...new Set(glosses.flatMap((g) => g.replace(/\([^)]*\)/g, '').split(/[,;]/)) + .map((s) => s.trim().toLowerCase().replace(/^(to|a|an|the) /, '')).filter((s) => /^[a-z][a-z' -]{1,30}$/.test(s) && s.split(' ').length <= 3))]; + +// ur.wiktionary Urdu entries: numbered lines under ==معانی==, else "# " lines, else the opening prose; +// origin from "(عربی)" on the first line +export function urduEntry(wt: string) { + const m = wt.split(/==\s*معانی\s*==/)[1]?.split(/\n==/)[0] ?? ''; + let defs = m.split('\n').map((l) => l.match(/^\s*(?:\d+[.)-]|#)\s*(.+)/)?.[1]).filter(Boolean).map((l) => unlink(l!)); + if (!defs.length) defs = definitions(wt, 'ur').defs; + if (!defs.length) { + const prose = wt.split('\n').find((l) => l.trim().startsWith("'''") && l.length > 20); + if (prose) defs = [unlink(stripTemplates(prose)).slice(0, 300)]; + } + const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null; + const english = wt.match(/^\s*انگریزی\s*:\s*([A-Za-z].+)$/m)?.[1].trim() ?? null; // "== تراجم ==" pages + return { defs: defs.slice(0, 6), origin, links: [] as string[], ...(english && { english }) }; +} + +// ur.wiktionary English entries ("north": "# [[شمالی]]۔"): the Urdu words, for the pivot through English +export const isEnglish = (title: string) => /^[A-Za-z][A-Za-z' -]*$/.test(title); +export function englishToUrdu(wt: string) { + const lines = wt.split('\n').filter((l) => /^#(?![:*])/.test(l)).join(' '); + return [...new Set([...lines.matchAll(/\[\[(?:[^\]|]*\|)?([^\]]+)\]\]/g)].map((m) => m[1].trim()) + .filter((w) => /^[\p{Script=Arabic}\s]+$/u.test(w)))].slice(0, 10); +} +const stripTemplates = (t: string) => { + while (/\{\{[^{}]*\}\}/.test(t)) t = t.replace(/\{\{[^{}]*\}\}/g, ''); + return t; +}; + +// fa./ar.wiktionary: "# definition" lines outside etymology/translation sections; on ar.wiktionary only the +// Arabic section, and undiacritised pages that just point to entries ("* [[عِشْق]]") give those links export function definitions(wt: string, code: string) { if (code === 'ar' && wt.includes('{{اللغة|')) wt = wt.split(/==\s*\{\{اللغة\|/).find((x) => x.startsWith('عربية')) ?? ''; const defs: string[] = [], links: string[] = []; @@ -55,102 +96,214 @@ export function definitions(wt: string, code: string) { if (only) { links.push(only[1]); continue; } const m = line.match(/^#(?![:*])\s*(.+)/); if (!m) continue; - let t = m[1]; - while (/\{\{[^{}]*\}\}/.test(t)) t = t.replace(/\{\{[^{}]*\}\}/g, ''); - t = unlink(t).replace(/^[\s.،:-]+|[\s]+$/g, ''); + const t = unlink(stripTemplates(m[1])).replace(/^[\s.،:-]+|\s+$/g, ''); if (t.length >= 3 && !/^-+$/.test(t)) defs.push(t); } - return { defs: defs.slice(0, 5), links }; + return { defs: defs.slice(0, 6), origin: null as string | null, links: links.slice(0, 5) }; } -// meanings from the language's own Wiktionary: the first spelling with an entry (following one pointer hop) -async function native(code: string, word: string) { - const host = `${code}.wiktionary.org`; - for (const f of formsIn(code, word)) { - const wt = await wikitext(host, f); - if (!wt) continue; - if (code === 'ur') { - const e = urduEntry(wt); - if (e.meanings.length) return { defs: e.meanings, origin: e.origin, url: `https://${host}/wiki/${encodeURIComponent(f)}` }; - continue; +export const ownEntry = (code: string, wt: string) => (code === 'ur' ? urduEntry(wt) : definitions(wt, code)); + +// pages of a MediaWiki XML dump (articles only, no redirects) +export function* dumpPages(xml: string) { + for (const m of xml.matchAll(/([\s\S]*?)<\/page>/g)) { + const p = m[1]; + if (!/0<\/ns>/.test(p) || /([^<]*)<\/title>/)?.[1]; + const text = p.match(/]*>([\s\S]*?)<\/text>/)?.[1]; + if (title && text) yield { title: xmlText(title), text: xmlText(text) }; + } +} +const xmlText = (s: string) => s.replace(/</g, '<').replace(/>/g, '>').replace(/"/g, '"').replace(/'/g, "'").replace(/&/g, '&'); + +// ---- import and sync ---- + +const meta = async (k: string) => (await pool.query('SELECT value FROM dict_meta WHERE name = $1', [k])).rows[0]?.value ?? null; +const setMeta = (k: string, v: string) => + pool.query('INSERT INTO dict_meta (name, value) VALUES ($1, $2) ON CONFLICT (name) DO UPDATE SET value = $2', [k, v]); + +async function insertRows(client: any, rows: unknown[][]) { + for (let i = 0; i < rows.length; i += 500) { + const chunk = rows.slice(i, i + 500), values: unknown[] = []; + const sql = chunk.map((r, j) => { values.push(...r); const o = j * 5; return `($${o + 1},$${o + 2},$${o + 3},$${o + 4},$${o + 5})`; }); + await client.query(`INSERT INTO wiktionary (lang, source, title, key, data) VALUES ${sql.join(',')}`, values); + } +} + +// replace one source's rows in a transaction (readers see the old data until it commits) +async function replace(lang: string, source: string, fill: (add: (title: string, data: object) => Promise, client: any) => Promise) { + const client = await pool.connect(); + let n = 0; + try { + await client.query('BEGIN'); + await client.query('DELETE FROM wiktionary WHERE lang = $1 AND source = $2', [lang, source]); + if (lang === 'ur') await client.query('DELETE FROM ur_glosses WHERE source = $1', [source]); + let batch: unknown[][] = []; + await fill(async (title, data) => { + batch.push([lang, source, title, key(title), data]); n++; + if (batch.length >= 2000) { await insertRows(client, batch); batch = []; } + }, client); + await insertRows(client, batch); + await client.query('COMMIT'); + } catch (e) { + await client.query('ROLLBACK'); + throw e; + } finally { + client.release(); + } + return n; +} + +async function lastModified(url: string) { + const res = await fetch(url, { method: 'HEAD', headers: { 'user-agent': UA } }); + if (!res.ok) throw new Error(`${res.status} ${url}`); + return res.headers.get('last-modified') ?? ''; +} + +async function importKaikki(lang: string) { + const res = await fetch(KAIKKI(lang), { headers: { 'user-agent': UA } }); + if (!res.ok || !res.body) throw new Error(`${res.status} ${KAIKKI(lang)}`); + return replace(lang, 'en', async (add, client) => { + const glossRows: string[][] = []; + for await (const line of createInterface({ input: Readable.fromWeb(res.body as any), crlfDelay: Infinity })) { + if (!line) continue; + const d = JSON.parse(line), e = kaikkiEntry(d); + if (!e.glosses.length && !e.ipa.length) continue; + await add(d.word, e); + if (lang === 'ur' && !e.formOf) for (const g of glossKeys(e.glosses.slice(0, 3))) glossRows.push([g, d.word]); } - let { defs, links } = definitions(wt, code), title = f; - if (!defs.length && links[0]) ({ defs } = definitions(await wikitext(host, (title = links[0])), code)); - if (defs.length) return { defs, origin: null, url: `https://${host}/wiki/${encodeURIComponent(title)}` }; + await insertGlosses(client, 'en', glossRows); + }); +} + +async function insertGlosses(client: any, source: string, rows: string[][]) { + for (let i = 0; i < rows.length; i += 1000) { + const chunk = rows.slice(i, i + 1000); + await client.query(`INSERT INTO ur_glosses (gloss, word, source) VALUES ${chunk.map((_, j) => `($${j * 2 + 1},$${j * 2 + 2},'${source}')`).join(',')}`, chunk.flat()); } - return null; } -// ur.wiktionary: numbered lines under ==معانی==, origin from "(عربی)" on the first line -export function urduEntry(wt: string) { - const m = wt.split(/==\s*معانی\s*==/)[1]?.split(/\n==/)[0] ?? ''; - const meanings = m.split('\n').map((l) => l.match(/^\s*(?:\d+[.)-]|#)\s*(.+)/)?.[1]).filter(Boolean).map((l) => unlink(l!)).slice(0, 6); - const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null; - return { meanings, origin }; +async function importDump(lang: string) { + const bz = spawn('sh', ['-c', `curl -sfL -A '${UA}' '${DUMP(lang)}' | bzcat`]); + return replace(lang, 'own', async (add, client) => { + let buf = ''; + const glossRows: string[][] = []; // ur.wiktionary English entries -> the pivot table + for await (const chunk of bz.stdout.setEncoding('utf8')) { + buf += chunk; + const end = buf.lastIndexOf(''); + if (end < 0) continue; + for (const p of dumpPages(buf.slice(0, end + 7))) { + if (lang === 'ur' && isEnglish(p.title)) { + for (const w of englishToUrdu(p.text)) glossRows.push([p.title.toLowerCase(), w]); + continue; + } + const e = ownEntry(lang, p.text); + if (e.defs.length || e.links.length || 'english' in e) await add(p.title, e); + } + buf = buf.slice(end + 7); + } + const code: number = await new Promise((r) => (bz.exitCode !== null ? r(bz.exitCode) : bz.on('close', r))); + if (code !== 0) throw new Error(`dump download/decompress failed for ${lang} (exit ${code})`); + await insertGlosses(client, 'own', glossRows); + }); } -// en.wiktionary page HTML: per language, IPA, headword transliteration, audio -export function pronunciations(html: string) { - const out: Record = {}; - const parts = html.split(/]*id="([^"]+)"/); - for (let i = 1; i < parts.length; i += 2) { - const code = Object.entries(LANGS).find(([, n]) => n === parts[i])?.[0]; - if (!code) continue; - const body = parts[i + 1]; - const ipa = [...new Set([...body.matchAll(/class="IPA[^"]*"[^>]*>([^<]+)/g)].map((m) => m[1]).filter((x) => /^[/[]/.test(x)))].slice(0, 2); - const tr = body.match(new RegExp(`lang="${code}-Latn" class="headword-tr[^"]*"[^>]*>([^<]+)`))?.[1] ?? null; - const a = body.match(/"(?:https:)?(\/\/upload\.wikimedia\.org\/[^"]+\.(?:ogg|oga|mp3|wav))"/)?.[1]; - out[code] = { ipa, tr: tr && plain(tr), audio: a ? 'https:' + a : null }; +// changes on ur./fa./ar.wiktionary since the last run, re-read from the API (titles in batches of 50) +async function recentChanges(lang: string) { + const api = `https://${lang}.wiktionary.org/w/api.php`; + const since = await meta(`rc:${lang}`), now = new Date().toISOString(); + if (!since) return setMeta(`rc:${lang}`, now).then(() => 0); // first run: the dump is the baseline + const titles = new Set(); + let cont: Record = {}; + do { + const q = new URLSearchParams({ action: 'query', list: 'recentchanges', rcnamespace: '0', rctype: 'edit|new', rcprop: 'title', + rclimit: '500', rcdir: 'newer', rcstart: since, rcend: now, format: 'json', formatversion: '2', ...cont }); + const d: any = await (await fetch(`${api}?${q}`, { headers: { 'user-agent': UA } })).json(); + for (const c of d.query?.recentchanges ?? []) titles.add(c.title); + cont = d.continue ?? {}; + } while (cont.rccontinue); + const list = [...titles]; + for (let i = 0; i < list.length; i += 50) { + const q = new URLSearchParams({ action: 'query', prop: 'revisions', rvprop: 'content', rvslots: 'main', + titles: list.slice(i, i + 50).join('|'), format: 'json', formatversion: '2' }); + const d: any = await (await fetch(api, { method: 'POST', body: q, headers: { 'user-agent': UA } })).json(); + for (const p of d.query?.pages ?? []) { + const wt = p.revisions?.[0]?.slots?.main?.content; + if (lang === 'ur' && isEnglish(p.title)) { + const g = p.title.toLowerCase(); + await pool.query(`DELETE FROM ur_glosses WHERE source = 'own' AND gloss = $1`, [g]); + for (const w of wt ? englishToUrdu(wt) : []) await pool.query(`INSERT INTO ur_glosses (gloss, word, source) VALUES ($1, $2, 'own')`, [g, w]); + continue; + } + await pool.query(`DELETE FROM wiktionary WHERE lang = $1 AND source = 'own' AND title = $2`, [lang, p.title]); + const e = wt && !/^#(REDIRECT|تحويل|تغییر)/i.test(wt) ? ownEntry(lang, wt) : null; + if (e && (e.defs.length || e.links.length || 'english' in e)) + await pool.query(`INSERT INTO wiktionary (lang, source, title, key, data) VALUES ($1, 'own', $2, $3, $4)`, [lang, p.title, key(p.title), e]); + } } - return out; + await setMeta(`rc:${lang}`, now); + return list.length; } -// Urdu equivalents of single-word English glosses, from their translation tables -async function pivot(word: string, glosses: string[]) { - const seen = new Set([word.replace(DIAC, '')]); // the word itself, and duplicates differing only in diacritics - const words = [...new Set(glosses.flatMap((g) => g.split(/[,;]/)).map((s) => s.trim()).filter((s) => /^[a-z]+$/.test(s)))].slice(0, 3); // lowercase: no proper nouns - const found = await Promise.all(words.map(async (gloss) => { - const wt = await wikitext('en.wiktionary.org', gloss); - const lines = [...wt.matchAll(/^\*:? Urdu: (.*)$/gm)].map((m) => m[1]).join(' '); - const urdu = [...lines.matchAll(/\{\{t\+?\|ur\|([^|}]+)/g)].map((m) => m[1].trim()) - .filter((u) => !seen.has(u.replace(DIAC, '')) && seen.add(u.replace(DIAC, ''))).slice(0, 8); - return { gloss, urdu }; - })); - return found.filter((f) => f.urdu.length); +// full re-import of each source whose upstream file changed, then the daily recent changes +export async function sync(log = console.log) { + for (const lang of LANGS) { + for (const [source, url, run] of [['en', KAIKKI(lang), importKaikki], ['own', DUMP(lang), importDump]] as const) { + const lm = await lastModified(url), k = `file:${lang}:${source}`; + if (lm && lm === (await meta(k))) continue; + const t = Date.now(), n = await run(lang); + await setMeta(k, lm); + if (source === 'own') await setMeta(`rc:${lang}`, new Date(lm).toISOString()); // changes after the dump + log(`${lang} ${source}: ${n} entries (${Math.round((Date.now() - t) / 1000)} s)`); + } + log(`${lang} recent changes: ${await recentChanges(lang)} pages`); + } } -async function fetchWord(word: string) { - let title: string | null = null, defs: any = null; - for (const f of forms(word)) if ((defs = await get(rest('definition', f)))) { title = f; break; } - const [pron, ...own] = await Promise.all([ - title ? get(rest('html', title), false).then((h) => pronunciations(h ?? '')) : {}, - ...Object.keys(LANGS).map((c) => native(c, word)), - ]) as [Record, ...(Awaited>)[]]; - const langs = Object.keys(LANGS).map((code, i) => ({ - code, ...(pron[code] ?? { ipa: [], tr: null, audio: null }), - meanings: own[i]?.defs ?? [], origin: own[i]?.origin ?? null, source: own[i]?.url ?? null, - senses: (defs?.[code] ?? []).map((e: any) => ({ - pos: e.partOfSpeech, defs: e.definitions.map((d: any) => plain(d.definition)).filter(Boolean).slice(0, 4), - })).filter((s: any) => s.defs.length).slice(0, 3), - })).filter((l) => l.meanings.length || l.senses.length || l.ipa.length || l.tr); - const glosses = langs.find((l) => l.senses.length)?.senses[0].defs.slice(0, 2) ?? []; // the primary meaning only - const hasUrdu = langs.some((l) => l.code === 'ur' && l.meanings.length); - return { - word, langs, - equivalents: hasUrdu ? [] : await pivot(word, glosses), - sources: { en: title && `https://en.wiktionary.org/wiki/${encodeURIComponent(title)}` }, - found: langs.length > 0, - }; -} +// ---- lookup ---- + +const page = (lang: string, source: string, title: string) => + source === 'en' ? `https://en.wiktionary.org/wiki/${encodeURIComponent(title)}#${NAME[lang as 'ur']}` : `https://${lang}.wiktionary.org/wiki/${encodeURIComponent(title)}`; export async function lookup(word: string) { - const hit = await pool.query(`SELECT data FROM dictionary WHERE word = $1 AND fetched_at > now() - interval '30 days'`, [word]); - if (hit.rows[0]) return hit.rows[0].data; - const data = await fetchWord(word); - if (!data.found) return data; // not found (or Wiktionary unreachable): don't cache - await pool.query( - `INSERT INTO dictionary (word, data) VALUES ($1, $2) ON CONFLICT (word) DO UPDATE SET data = $2, fetched_at = now()`, - [word, data], - ); - return data; + const k = key(word); + const { rows } = await pool.query('SELECT lang, source, title, data FROM wiktionary WHERE key = $1 ORDER BY id', [k]); + // ar.wiktionary pointer pages: follow to the diacritised entries + const pointers = rows.filter((r) => r.source === 'own' && !r.data.defs.length).flatMap((r) => r.data.links.map((l: string) => [r.lang, l])); + if (pointers.length) { + const more = await pool.query( + `SELECT lang, source, title, data FROM wiktionary WHERE source = 'own' AND (lang, title) IN (SELECT * FROM unnest($1::text[], $2::text[]))`, + [pointers.map((p) => p[0]), pointers.map((p) => p[1])]); + rows.push(...more.rows); + } + const langs = LANGS.map((code) => { + const en = rows.filter((r) => r.lang === code && r.source === 'en'); + const own = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.defs.length); + const lemmas = en.filter((r) => !r.data.formOf), use = lemmas.length ? lemmas : en; + const first = (f: string) => use.map((r) => r.data[f]).find((v) => (Array.isArray(v) ? v.length : v)) ?? null; + // ur.wiktionary's English translation lines ("انگریزی : …") + const translated = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.english).map((r) => r.data.english); + return { + code, ipa: first('ipa') ?? [], tr: first('tr'), audio: first('audio'), ety: first('ety'), + synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))].slice(0, 8), + meanings: own.flatMap((r) => r.data.defs).slice(0, 6), origin: own.find((r) => r.data.origin)?.data.origin ?? null, + source: own[0] ? page(code, 'own', own[0].title) : null, + senses: [...use.map((r) => ({ pos: r.data.pos, defs: r.data.glosses.slice(0, 4) })).filter((s) => s.defs.length).slice(0, 3), + ...(translated.length ? [{ pos: 'translation', defs: translated }] : [])], + en: en[0] ? page(code, 'en', en[0].title) : null, + }; + }).filter((l) => l.meanings.length || l.senses.length || l.ipa.length || l.tr); + // no Urdu meaning: Urdu words sharing the primary English meaning + let equivalents: { gloss: string; urdu: string[] }[] = []; + if (!langs.some((l) => l.code === 'ur' && l.meanings.length)) { + const glosses = glossKeys(langs.find((l) => l.senses.length)?.senses[0].defs.slice(0, 2) ?? []).slice(0, 4); + if (glosses.length) { + const { rows: eq } = await pool.query('SELECT gloss, word FROM ur_glosses WHERE gloss = ANY($1)', [glosses]); + const seen = new Set([k]); + equivalents = glosses.map((g) => ({ + gloss: g, urdu: eq.filter((r) => r.gloss === g && !seen.has(key(r.word)) && seen.add(key(r.word))).map((r) => r.word).slice(0, 8), + })).filter((e) => e.urdu.length); + } + } + return { word, langs, equivalents, found: langs.length > 0 }; } diff --git a/db/schema.sql b/db/schema.sql index dc20eff7..b9edef02 100644 --- a/db/schema.sql +++ b/db/schema.sql @@ -58,9 +58,25 @@ CREATE TABLE IF NOT EXISTS verses ( PRIMARY KEY (poem_id, vorder) ); --- Wiktionary lookups for the reading sidebar (api/src/dictionary.ts), refreshed after 30 days -CREATE TABLE IF NOT EXISTS dictionary ( - word text PRIMARY KEY, - data jsonb NOT NULL, - fetched_at timestamptz NOT NULL DEFAULT now() +-- Wiktionary for the word sidebar (api/src/dictionary.ts; filled and kept current by dict-sync.ts) +DROP TABLE IF EXISTS dictionary; -- the earlier live-lookup cache +CREATE TABLE IF NOT EXISTS wiktionary ( + id bigserial PRIMARY KEY, + lang text NOT NULL, -- the word's language: ur | fa | ar + source text NOT NULL, -- en (en.wiktionary, via kaikki.org) | own (that language's Wiktionary) + title text NOT NULL, -- headword / page title + key text NOT NULL, -- spelling-insensitive lookup key (dictionary.ts key()) + data jsonb NOT NULL +); +CREATE INDEX IF NOT EXISTS wiktionary_key ON wiktionary(key); +CREATE INDEX IF NOT EXISTS wiktionary_title ON wiktionary(lang, source, title); +CREATE TABLE IF NOT EXISTS ur_glosses ( -- English gloss -> Urdu word, for the pivot through English + gloss text NOT NULL, + word text NOT NULL +); +ALTER TABLE ur_glosses ADD COLUMN IF NOT EXISTS source text NOT NULL DEFAULT 'en'; -- en | own (ur.wiktionary English entries) +CREATE INDEX IF NOT EXISTS ur_glosses_gloss ON ur_glosses(gloss); +CREATE TABLE IF NOT EXISTS dict_meta ( -- upstream file versions and recent-changes timestamps + name text PRIMARY KEY, + value text NOT NULL ); diff --git a/deploy/sync.sh b/deploy/sync.sh index 8ede1c40..bfc01e8b 100755 --- a/deploy/sync.sh +++ b/deploy/sync.sh @@ -4,6 +4,7 @@ # 1. update the divan-data checkout and fetch new/edited works from Wikisource (incremental) # 2. rebuild its search index and site export # 3. load the export into PostgreSQL (upserts; the site reads it live) +# 3b. sync the word dictionary from Wiktionary (api/src/dict-sync.ts; first run imports ~700 MB) # 4. optionally commit + push the refreshed data (DIVAN_DATA_PUSH=1, needs git push access) # # Schedule with cron, e.g. daily at 03:15: @@ -11,7 +12,7 @@ # # Settings (env): DIVAN_DATA_DIR (default /opt/divan-data), DIVAN_APP_DIR (default: this repo), # DATABASE_URL (Postgres, as for the API), DIVAN_DATA_PUSH=1 to push data changes. -# Needs: git, python3 (stdlib only), node 24+. +# Needs: git, python3 (stdlib only), node 24+, curl and bzcat (dictionary dumps). set -euo pipefail APP_DIR=${DIVAN_APP_DIR:-$(cd "$(dirname "$0")/.." && pwd)} @@ -36,4 +37,8 @@ fi cd "$APP_DIR/api" [ -d node_modules ] || npm ci --omit=dev --silent node src/import.ts "$DATA_DIR" + +# word dictionary (Wiktionary): re-imports sources that changed upstream, then the day's edits. +# A failure here must not fail the content sync. +node src/dict-sync.ts || echo "$(date -u +%FT%TZ) dictionary sync failed" echo "== $(date -u +%FT%TZ) done" diff --git a/web/src/layouts/Base.astro b/web/src/layouts/Base.astro index 431b5999..45b696ee 100644 --- a/web/src/layouts/Base.astro +++ b/web/src/layouts/Base.astro @@ -120,7 +120,8 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک const dict = document.getElementById('dict'), dictBody = dict.querySelector('.dict-body'), main = document.querySelector('main'); const LANG = { ur: 'اردو', fa: 'فارسی', ar: 'عربی' }; const POS = { Noun: 'اسم', 'Proper noun': 'اسم معرفہ', Verb: 'فعل', Adjective: 'صفت', Adverb: 'متعلق فعل', Pronoun: 'ضمیر', - Postposition: 'حرف جار', Preposition: 'حرف جار', Conjunction: 'حرف عطف', Interjection: 'حرف ندا', Particle: 'حرف' }; + Postposition: 'حرف جار', Preposition: 'حرف جار', Conjunction: 'حرف عطف', Interjection: 'حرف ندا', Particle: 'حرف', + Translation: 'ترجمہ (اردو ویکی لغت)', Phrase: 'فقرہ', Proverb: 'کہاوت', Name: 'اسم معرفہ' }; const el = (tag, cls, text, attrs) => { const e = document.createElement(tag); if (cls) e.className = cls; @@ -134,7 +135,8 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک current = w; dict.hidden = false; document.body.classList.add('dict-open'); dictBody.replaceChildren(el('h2', null, w), el('p', 'muted', 'تلاش جاری ہے…')); let d = null; - try { const r = await fetch('/api/word?w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {} + // v: bump when the response format changes (responses are browser-cached for a day) + try { const r = await fetch('/api/word?v=2&w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {} if (w !== current) return; // a newer selection won const out = [el('h2', null, w)]; if (!d || !d.found) { @@ -171,14 +173,20 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک if (l.source) label.append(el('a', null, 'ماخذ', { href: l.source, target: '_blank', rel: 'noopener' })); sec.append(label, list(l.meanings, null, { lang: l.code })); } - for (const s of l.senses) sec.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' })); + for (const s of l.senses) sec.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos.charAt(0).toUpperCase() + s.pos.slice(1)] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' })); + if (l.synonyms.length) sec.append(el('p', 'pos', 'مترادفات'), el('p', null, l.synonyms.join('، '))); + if (l.ety) sec.append(el('p', 'pos', 'اشتقاق'), el('p', 'en ety', l.ety, { dir: 'ltr', lang: 'en' })); + if (l.en) { + const p = el('p', 'pos'); + p.append(el('a', null, 'انگریزی ویکی لغت', { href: l.en, target: '_blank', rel: 'noopener' })); + sec.append(p); + } out.push(sec); if (l.code === 'ur' && d.equivalents.length) out.push(equivalents()); } if (d.equivalents.length && !d.langs.some((l) => l.code === 'ur')) out.splice(1, 0, equivalents()); const src = el('p', 'src muted'); - src.append('ویکی لغت · '); - if (d.sources.en) src.append(el('a', null, 'انگریزی ویکی لغت', { href: d.sources.en, target: '_blank', rel: 'noopener' }), ' · '); + src.append('ویکی لغت (Wiktionary) · '); src.append('CC BY-SA'); dictBody.replaceChildren(...out, src); } diff --git a/web/src/styles/global.css b/web/src/styles/global.css index ab19cca3..7e3e768d 100644 --- a/web/src/styles/global.css +++ b/web/src/styles/global.css @@ -146,6 +146,7 @@ h1 + .muted { text-align: center; margin-top: 0; } .dict .pos { font-size: .8rem; color: var(--muted); margin: 8px 0 0; } .dict .note { font-size: .8rem; margin: 0; } .dict .src { font-size: .78rem; margin-top: 18px; } +.dict .ety { text-align: left; color: var(--muted); font-size: .8rem; } .dict-close { position: sticky; top: 0; float: left; border: 0; background: none; color: var(--muted); font-size: 1rem; cursor: pointer; } .dict .play { border: 1px solid var(--border); background: var(--inner); border-radius: 8px; cursor: pointer; padding: 0 6px; } /* wide screens: the page moves over instead of being covered */