From ef618094d5e9d1903d9cf9d06dd9f6282bc4c5c2 Mon Sep 17 00:00:00 2001 From: Anas Rashid Date: Thu, 8 Oct 2026 21:38:10 +0200 Subject: [PATCH] Word sidebar: select a word for its meanings and pronunciation from Wiktionary Urdu, Persian and Arabic Wiktionaries in that order, each labelled; English meanings, IPA, transliteration and audio from en.wiktionary; Urdu equivalents pivoted through English when Urdu Wiktionary has no entry. Cached in PostgreSQL for 30 days. Left panel on wide screens, bottom sheet on phones. Co-Authored-By: Claude Opus 5.5 --- README.md | 1 + api/src/dictionary.test.ts | 34 ++++++++ api/src/dictionary.ts | 156 ++++++++++++++++++++++++++++++++++ api/src/server.ts | 9 ++ db/schema.sql | 7 ++ web/src/layouts/Base.astro | 85 ++++++++++++++++++ web/src/pages/[...path].astro | 1 + web/src/pages/api/word.ts | 12 +++ web/src/styles/global.css | 24 ++++++ 9 files changed, 329 insertions(+) create mode 100644 api/src/dictionary.test.ts create mode 100644 api/src/dictionary.ts create mode 100644 web/src/pages/api/word.ts diff --git a/README.md b/README.md index 933dc6f7..9208d68a 100644 --- a/README.md +++ b/README.md @@ -51,6 +51,7 @@ Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DI | `GET /api/poets` | all poets | | `GET /api/page?url=/p238/...` | the poet, category or poem at a site URL (breadcrumbs, children, verses, prev/next) | | `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised; exact phrase first; each with the best-matching couplet or paragraph (`snippet`) | +| `GET /api/word?w=` | one word's meanings and pronunciation from Wiktionary (Urdu, Persian, Arabic in that order; English meanings; Urdu equivalents via English when Urdu Wiktionary has none), cached 30 days | | `GET /health` | database check | Search normalises both stored text and queries: Arabic ي/ك/ه → Urdu ی/ک/ہ, ۂ/ۓ, diacritics and the Urdu full stop removed; do-chashmi ھ stays distinct. diff --git a/api/src/dictionary.test.ts b/api/src/dictionary.test.ts new file mode 100644 index 00000000..4036aa44 --- /dev/null +++ b/api/src/dictionary.test.ts @@ -0,0 +1,34 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { forms, urduEntry, pronunciations } from './dictionary.ts'; + +test('spelling forms: diacritics dropped, final noon ghunna tried as noon', () => { + assert.deepEqual(forms('ناداں'), ['ناداں', 'نادان']); + assert.deepEqual(forms('وِصال'), ['وِصال', 'وصال']); +}); + +test('ur.wiktionary entry: meanings and origin', () => { + const wt = "وِصال {وِصال} ([[عربی]])\n\nاسم [[نکرہ]]\n\n==معانی==\n\n1. بھیٹ، ملاقات، [[عاشق و معشوق]] کی محبت۔\n\n==مرکبات==\n\nوِصال کا دِن"; + assert.deepEqual(urduEntry(wt), { meanings: ['بھیٹ، ملاقات، عاشق و معشوق کی محبت۔'], origin: 'عربی' }); +}); + +test('en.wiktionary HTML: IPA, transliteration and audio per language', () => { + const html = '

Persian

/qis.ˈmat/-at' + + 'qismat' + + '

Urdu

/qɪs.mət̪/qismat'; + assert.deepEqual(pronunciations(html), { + fa: { ipa: ['/qis.ˈmat/'], tr: 'qismat', audio: 'https://upload.wikimedia.org/a/b.ogg' }, + ur: { ipa: ['/qɪs.mət̪/'], tr: 'qismat', audio: null }, + }); +}); + +test('fa./ar.wiktionary definitions: skip etymology, follow Arabic pointers', async () => { + const { definitions, formsIn } = await import('./dictionary.ts'); + const fa = '==فارسی==\n===ریشه‌شناسی===\n* [[عربی]]\n# قسمة\n===اسم===\n#بهره، نصیب.\n#:example\n#سرنوشت، تقدیر.\n====برگردان‌ها====\n# x y z'; + assert.deepEqual(definitions(fa, 'fa').defs, ['بهره، نصیب.', 'سرنوشت، تقدیر.']); + const ar = '== {{اللغة|عربية}} ==\nهل تقصد:\n* [[عِشْق]]\n* [[عَشَقَ]]\n----\n== {{اللغة|أردية}} ==\n# [[x]]'; + assert.deepEqual(definitions(ar, 'ar'), { defs: [], links: ['عِشْق', 'عَشَقَ'] }); + assert.deepEqual(definitions('# {{مصدر|عَشِقَ}}.\n# فرط الحب.', 'ar').defs, ['فرط الحب.']); + assert.deepEqual(formsIn('fa', 'نگاہ'), ['نگاه']); + assert.deepEqual(formsIn('ar', 'معنی'), ['معني', 'معنى']); +}); diff --git a/api/src/dictionary.ts b/api/src/dictionary.ts new file mode 100644 index 00000000..e38240f1 --- /dev/null +++ b/api/src/dictionary.ts @@ -0,0 +1,156 @@ +// Word lookup for the reading sidebar, from Wiktionary (CC BY-SA). For Urdu, Persian and Arabic, in that order: +// - meanings in the language itself from its own Wiktionary (ur., fa., ar.wiktionary) +// - English meanings and pronunciation (IPA, transliteration, audio) from en.wiktionary +// With no Urdu meanings, Urdu equivalents are pivoted through English (the English glosses' translation tables). +// Found words are cached in PostgreSQL (dictionary table) for 30 days. +import { pool } from './db.ts'; + +const UA = 'Divan/2 (https://github.com/anas-rashid/divan)'; +const LANGS = { ur: 'Urdu', fa: 'Persian', ar: 'Arabic' } as const; +const DIAC = /[ً-ْٰٔٗ٘]/g; +const plain = (html: string) => + html.replace(/<[^>]+>/g, '').replace(/ /g, ' ').replace(/&/g, '&').replace(/&#\d+;/g, '').replace(/\s+/g, ' ').trim(); +const unlink = (wt: string) => wt.replace(/\[\[(?:[^\]|]*\|)?([^\]]*)\]\]/g, '$1').replace(/'''?/g, '').trim(); + +async function get(url: string, json = true): Promise { + const res = await fetch(url, { headers: { 'user-agent': UA }, signal: AbortSignal.timeout(8000) }).catch(() => null); + if (!res?.ok) return null; + return json ? res.json() : res.text(); +} +const rest = (path: string, title: string) => `https://en.wiktionary.org/api/rest_v1/page/${path}/${encodeURIComponent(title)}`; +const wikitext = async (host: string, title: string): Promise => { + const d = await get(`https://${host}/w/api.php?action=parse&page=${encodeURIComponent(title)}&prop=wikitext&format=json&formatversion=2&redirects=1`); + return d?.parse?.wikitext ?? ''; +}; + +// spelling forms to try: as written, without diacritics, final noon ghunna as noon (ناداں -> نادان) +export const forms = (w: string) => { + const bare = w.replace(DIAC, '').replace(/ۂ/g, 'ہ'); + return [...new Set([w, bare, bare.replace(/ں$/, 'ن')])]; +}; + +// the word in Persian / Arabic spelling (ہ -> ه, ی -> ي, ک -> ك …); Arabic also tries final alef maqsura +const SPELL: Record = { + fa: [[/[\u06C1\u06C2\u06BE]/g, '\u0647'], [/[\u06D2\u064A]/g, '\u06CC'], [/\u0643/g, '\u06A9']], + ar: [[/[\u06C1\u06BE]/g, '\u0647'], [/\u06C2/g, '\u0629'], [/[\u06CC\u06D2]/g, '\u064A'], [/\u06A9/g, '\u0643']], +}; +export function formsIn(code: string, w: string) { + const base = forms(w); + if (!SPELL[code]) return base; + const conv = base.map((f) => SPELL[code].reduce((s, [re, to]) => s.replace(re, to), f)); + return [...new Set([...conv, ...(code === 'ar' ? conv.map((f) => f.replace(/\u064A$/, '\u0649')) : [])])]; +} + +// fa./ar.wiktionary: "# definition" lines outside etymology/translation sections; on ar.wiktionary only +// the Arabic section, and undiacritised pages that just point to entries ("* [[عِشْق]]") give those links +export function definitions(wt: string, code: string) { + if (code === 'ar' && wt.includes('{{اللغة|')) wt = wt.split(/==\s*\{\{اللغة\|/).find((x) => x.startsWith('عربية')) ?? ''; + const defs: string[] = [], links: string[] = []; + let skip = false; + for (const line of wt.split('\n')) { + const h = line.match(/^=+\s*(.*?)\s*=+\s*$/); + if (h) { skip = /ریشه|ترجم|برگردان|تصريف|تصریف|منابع|مشتق|نفس الجذر/.test(h[1]); continue; } + if (skip) continue; + const only = line.match(/^[#*]\s*'*\[\[([^\]|]+)\]\]'*\s*(?:\([^)]*\))?\.?\s*$/); + if (only) { links.push(only[1]); continue; } + const m = line.match(/^#(?![:*])\s*(.+)/); + if (!m) continue; + let t = m[1]; + while (/\{\{[^{}]*\}\}/.test(t)) t = t.replace(/\{\{[^{}]*\}\}/g, ''); + t = unlink(t).replace(/^[\s.،:-]+|[\s]+$/g, ''); + if (t.length >= 3 && !/^-+$/.test(t)) defs.push(t); + } + return { defs: defs.slice(0, 5), links }; +} + +// meanings from the language's own Wiktionary: the first spelling with an entry (following one pointer hop) +async function native(code: string, word: string) { + const host = `${code}.wiktionary.org`; + for (const f of formsIn(code, word)) { + const wt = await wikitext(host, f); + if (!wt) continue; + if (code === 'ur') { + const e = urduEntry(wt); + if (e.meanings.length) return { defs: e.meanings, origin: e.origin, url: `https://${host}/wiki/${encodeURIComponent(f)}` }; + continue; + } + let { defs, links } = definitions(wt, code), title = f; + if (!defs.length && links[0]) ({ defs } = definitions(await wikitext(host, (title = links[0])), code)); + if (defs.length) return { defs, origin: null, url: `https://${host}/wiki/${encodeURIComponent(title)}` }; + } + return null; +} + +// ur.wiktionary: numbered lines under ==معانی==, origin from "(عربی)" on the first line +export function urduEntry(wt: string) { + const m = wt.split(/==\s*معانی\s*==/)[1]?.split(/\n==/)[0] ?? ''; + const meanings = m.split('\n').map((l) => l.match(/^\s*(?:\d+[.)-]|#)\s*(.+)/)?.[1]).filter(Boolean).map((l) => unlink(l!)).slice(0, 6); + const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null; + return { meanings, origin }; +} + +// en.wiktionary page HTML: per language, IPA, headword transliteration, audio +export function pronunciations(html: string) { + const out: Record = {}; + const parts = html.split(/]*id="([^"]+)"/); + for (let i = 1; i < parts.length; i += 2) { + const code = Object.entries(LANGS).find(([, n]) => n === parts[i])?.[0]; + if (!code) continue; + const body = parts[i + 1]; + const ipa = [...new Set([...body.matchAll(/class="IPA[^"]*"[^>]*>([^<]+)/g)].map((m) => m[1]).filter((x) => /^[/[]/.test(x)))].slice(0, 2); + const tr = body.match(new RegExp(`lang="${code}-Latn" class="headword-tr[^"]*"[^>]*>([^<]+)`))?.[1] ?? null; + const a = body.match(/"(?:https:)?(\/\/upload\.wikimedia\.org\/[^"]+\.(?:ogg|oga|mp3|wav))"/)?.[1]; + out[code] = { ipa, tr: tr && plain(tr), audio: a ? 'https:' + a : null }; + } + return out; +} + +// Urdu equivalents of single-word English glosses, from their translation tables +async function pivot(word: string, glosses: string[]) { + const seen = new Set([word.replace(DIAC, '')]); // the word itself, and duplicates differing only in diacritics + const words = [...new Set(glosses.flatMap((g) => g.split(/[,;]/)).map((s) => s.trim()).filter((s) => /^[a-z]+$/.test(s)))].slice(0, 3); // lowercase: no proper nouns + const found = await Promise.all(words.map(async (gloss) => { + const wt = await wikitext('en.wiktionary.org', gloss); + const lines = [...wt.matchAll(/^\*:? Urdu: (.*)$/gm)].map((m) => m[1]).join(' '); + const urdu = [...lines.matchAll(/\{\{t\+?\|ur\|([^|}]+)/g)].map((m) => m[1].trim()) + .filter((u) => !seen.has(u.replace(DIAC, '')) && seen.add(u.replace(DIAC, ''))).slice(0, 8); + return { gloss, urdu }; + })); + return found.filter((f) => f.urdu.length); +} + +async function fetchWord(word: string) { + let title: string | null = null, defs: any = null; + for (const f of forms(word)) if ((defs = await get(rest('definition', f)))) { title = f; break; } + const [pron, ...own] = await Promise.all([ + title ? get(rest('html', title), false).then((h) => pronunciations(h ?? '')) : {}, + ...Object.keys(LANGS).map((c) => native(c, word)), + ]) as [Record, ...(Awaited>)[]]; + const langs = Object.keys(LANGS).map((code, i) => ({ + code, ...(pron[code] ?? { ipa: [], tr: null, audio: null }), + meanings: own[i]?.defs ?? [], origin: own[i]?.origin ?? null, source: own[i]?.url ?? null, + senses: (defs?.[code] ?? []).map((e: any) => ({ + pos: e.partOfSpeech, defs: e.definitions.map((d: any) => plain(d.definition)).filter(Boolean).slice(0, 4), + })).filter((s: any) => s.defs.length).slice(0, 3), + })).filter((l) => l.meanings.length || l.senses.length || l.ipa.length || l.tr); + const glosses = langs.find((l) => l.senses.length)?.senses[0].defs.slice(0, 2) ?? []; // the primary meaning only + const hasUrdu = langs.some((l) => l.code === 'ur' && l.meanings.length); + return { + word, langs, + equivalents: hasUrdu ? [] : await pivot(word, glosses), + sources: { en: title && `https://en.wiktionary.org/wiki/${encodeURIComponent(title)}` }, + found: langs.length > 0, + }; +} + +export async function lookup(word: string) { + const hit = await pool.query(`SELECT data FROM dictionary WHERE word = $1 AND fetched_at > now() - interval '30 days'`, [word]); + if (hit.rows[0]) return hit.rows[0].data; + const data = await fetchWord(word); + if (!data.found) return data; // not found (or Wiktionary unreachable): don't cache + await pool.query( + `INSERT INTO dictionary (word, data) VALUES ($1, $2) ON CONFLICT (word) DO UPDATE SET data = $2, fetched_at = now()`, + [word, data], + ); + return data; +} diff --git a/api/src/server.ts b/api/src/server.ts index f8e0ce77..eb72d91b 100644 --- a/api/src/server.ts +++ b/api/src/server.ts @@ -2,10 +2,12 @@ // GET /api/poets all poets // GET /api/page?url=/p238/... poet, category or poem at that URL // GET /api/search?q=&poet=&page= +// GET /api/word?w= Wiktionary meanings and pronunciation (sidebar) // GET /health import Fastify from 'fastify'; import { pool } from './db.ts'; import { likePatterns, normalise, terms } from './urdu.ts'; +import { lookup } from './dictionary.ts'; const app = Fastify({ logger: { level: process.env.LOG_LEVEL ?? 'info' } }); const PAGE_SIZE = 20; @@ -124,5 +126,12 @@ app.get<{ Querystring: { q?: string; poet?: string; page?: string } }>('/api/sea return { total: count.rows[0].n, page, pageSize: PAGE_SIZE, results }; }); +// one word in Arabic script (Urdu, Persian, Arabic), as selected by a reader +app.get<{ Querystring: { w?: string } }>('/api/word', async (req, reply) => { + const w = (req.query.w ?? '').trim(); + if (!/^[\p{Script=Arabic}\p{M}\u200C]{1,40}$/u.test(w)) return reply.code(400).send({ error: 'one Urdu, Persian or Arabic word' }); + return lookup(w); +}); + const port = Number(process.env.PORT ?? 4100); await app.listen({ port, host: process.env.HOST ?? '127.0.0.1' }); diff --git a/db/schema.sql b/db/schema.sql index bbf4c6ed..dc20eff7 100644 --- a/db/schema.sql +++ b/db/schema.sql @@ -57,3 +57,10 @@ CREATE TABLE IF NOT EXISTS verses ( text text NOT NULL, PRIMARY KEY (poem_id, vorder) ); + +-- Wiktionary lookups for the reading sidebar (api/src/dictionary.ts), refreshed after 30 days +CREATE TABLE IF NOT EXISTS dictionary ( + word text PRIMARY KEY, + data jsonb NOT NULL, + fetched_at timestamptz NOT NULL DEFAULT now() +); diff --git a/web/src/layouts/Base.astro b/web/src/layouts/Base.astro index 9e7fb578..431b5999 100644 --- a/web/src/layouts/Base.astro +++ b/web/src/layouts/Base.astro @@ -48,6 +48,10 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
+