diff --git a/README.md b/README.md
index 933dc6f7..9208d68a 100644
--- a/README.md
+++ b/README.md
@@ -51,6 +51,7 @@ Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DI
| `GET /api/poets` | all poets |
| `GET /api/page?url=/p238/...` | the poet, category or poem at a site URL (breadcrumbs, children, verses, prev/next) |
| `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised; exact phrase first; each with the best-matching couplet or paragraph (`snippet`) |
+| `GET /api/word?w=` | one word's meanings and pronunciation from Wiktionary (Urdu, Persian, Arabic in that order; English meanings; Urdu equivalents via English when Urdu Wiktionary has none), cached 30 days |
| `GET /health` | database check |
Search normalises both stored text and queries: Arabic ي/ك/ه → Urdu ی/ک/ہ, ۂ/ۓ, diacritics and the Urdu full stop removed; do-chashmi ھ stays distinct.
diff --git a/api/src/dictionary.test.ts b/api/src/dictionary.test.ts
new file mode 100644
index 00000000..4036aa44
--- /dev/null
+++ b/api/src/dictionary.test.ts
@@ -0,0 +1,34 @@
+import { test } from 'node:test';
+import assert from 'node:assert/strict';
+import { forms, urduEntry, pronunciations } from './dictionary.ts';
+
+test('spelling forms: diacritics dropped, final noon ghunna tried as noon', () => {
+ assert.deepEqual(forms('ناداں'), ['ناداں', 'نادان']);
+ assert.deepEqual(forms('وِصال'), ['وِصال', 'وصال']);
+});
+
+test('ur.wiktionary entry: meanings and origin', () => {
+ const wt = "وِصال {وِصال} ([[عربی]])\n\nاسم [[نکرہ]]\n\n==معانی==\n\n1. بھیٹ، ملاقات، [[عاشق و معشوق]] کی محبت۔\n\n==مرکبات==\n\nوِصال کا دِن";
+ assert.deepEqual(urduEntry(wt), { meanings: ['بھیٹ، ملاقات، عاشق و معشوق کی محبت۔'], origin: 'عربی' });
+});
+
+test('en.wiktionary HTML: IPA, transliteration and audio per language', () => {
+ const html = '
Persian
/qis.ˈmat/-at' +
+ 'qismat' +
+ 'Urdu
/qɪs.mət̪/qismat';
+ assert.deepEqual(pronunciations(html), {
+ fa: { ipa: ['/qis.ˈmat/'], tr: 'qismat', audio: 'https://upload.wikimedia.org/a/b.ogg' },
+ ur: { ipa: ['/qɪs.mət̪/'], tr: 'qismat', audio: null },
+ });
+});
+
+test('fa./ar.wiktionary definitions: skip etymology, follow Arabic pointers', async () => {
+ const { definitions, formsIn } = await import('./dictionary.ts');
+ const fa = '==فارسی==\n===ریشهشناسی===\n* [[عربی]]\n# قسمة\n===اسم===\n#بهره، نصیب.\n#:example\n#سرنوشت، تقدیر.\n====برگردانها====\n# x y z';
+ assert.deepEqual(definitions(fa, 'fa').defs, ['بهره، نصیب.', 'سرنوشت، تقدیر.']);
+ const ar = '== {{اللغة|عربية}} ==\nهل تقصد:\n* [[عِشْق]]\n* [[عَشَقَ]]\n----\n== {{اللغة|أردية}} ==\n# [[x]]';
+ assert.deepEqual(definitions(ar, 'ar'), { defs: [], links: ['عِشْق', 'عَشَقَ'] });
+ assert.deepEqual(definitions('# {{مصدر|عَشِقَ}}.\n# فرط الحب.', 'ar').defs, ['فرط الحب.']);
+ assert.deepEqual(formsIn('fa', 'نگاہ'), ['نگاه']);
+ assert.deepEqual(formsIn('ar', 'معنی'), ['معني', 'معنى']);
+});
diff --git a/api/src/dictionary.ts b/api/src/dictionary.ts
new file mode 100644
index 00000000..e38240f1
--- /dev/null
+++ b/api/src/dictionary.ts
@@ -0,0 +1,156 @@
+// Word lookup for the reading sidebar, from Wiktionary (CC BY-SA). For Urdu, Persian and Arabic, in that order:
+// - meanings in the language itself from its own Wiktionary (ur., fa., ar.wiktionary)
+// - English meanings and pronunciation (IPA, transliteration, audio) from en.wiktionary
+// With no Urdu meanings, Urdu equivalents are pivoted through English (the English glosses' translation tables).
+// Found words are cached in PostgreSQL (dictionary table) for 30 days.
+import { pool } from './db.ts';
+
+const UA = 'Divan/2 (https://github.com/anas-rashid/divan)';
+const LANGS = { ur: 'Urdu', fa: 'Persian', ar: 'Arabic' } as const;
+const DIAC = /[ً-ْٰٔٗ٘]/g;
+const plain = (html: string) =>
+ html.replace(/<[^>]+>/g, '').replace(/ /g, ' ').replace(/&/g, '&').replace(/\d+;/g, '').replace(/\s+/g, ' ').trim();
+const unlink = (wt: string) => wt.replace(/\[\[(?:[^\]|]*\|)?([^\]]*)\]\]/g, '$1').replace(/'''?/g, '').trim();
+
+async function get(url: string, json = true): Promise {
+ const res = await fetch(url, { headers: { 'user-agent': UA }, signal: AbortSignal.timeout(8000) }).catch(() => null);
+ if (!res?.ok) return null;
+ return json ? res.json() : res.text();
+}
+const rest = (path: string, title: string) => `https://en.wiktionary.org/api/rest_v1/page/${path}/${encodeURIComponent(title)}`;
+const wikitext = async (host: string, title: string): Promise => {
+ const d = await get(`https://${host}/w/api.php?action=parse&page=${encodeURIComponent(title)}&prop=wikitext&format=json&formatversion=2&redirects=1`);
+ return d?.parse?.wikitext ?? '';
+};
+
+// spelling forms to try: as written, without diacritics, final noon ghunna as noon (ناداں -> نادان)
+export const forms = (w: string) => {
+ const bare = w.replace(DIAC, '').replace(/ۂ/g, 'ہ');
+ return [...new Set([w, bare, bare.replace(/ں$/, 'ن')])];
+};
+
+// the word in Persian / Arabic spelling (ہ -> ه, ی -> ي, ک -> ك …); Arabic also tries final alef maqsura
+const SPELL: Record = {
+ fa: [[/[\u06C1\u06C2\u06BE]/g, '\u0647'], [/[\u06D2\u064A]/g, '\u06CC'], [/\u0643/g, '\u06A9']],
+ ar: [[/[\u06C1\u06BE]/g, '\u0647'], [/\u06C2/g, '\u0629'], [/[\u06CC\u06D2]/g, '\u064A'], [/\u06A9/g, '\u0643']],
+};
+export function formsIn(code: string, w: string) {
+ const base = forms(w);
+ if (!SPELL[code]) return base;
+ const conv = base.map((f) => SPELL[code].reduce((s, [re, to]) => s.replace(re, to), f));
+ return [...new Set([...conv, ...(code === 'ar' ? conv.map((f) => f.replace(/\u064A$/, '\u0649')) : [])])];
+}
+
+// fa./ar.wiktionary: "# definition" lines outside etymology/translation sections; on ar.wiktionary only
+// the Arabic section, and undiacritised pages that just point to entries ("* [[عِشْق]]") give those links
+export function definitions(wt: string, code: string) {
+ if (code === 'ar' && wt.includes('{{اللغة|')) wt = wt.split(/==\s*\{\{اللغة\|/).find((x) => x.startsWith('عربية')) ?? '';
+ const defs: string[] = [], links: string[] = [];
+ let skip = false;
+ for (const line of wt.split('\n')) {
+ const h = line.match(/^=+\s*(.*?)\s*=+\s*$/);
+ if (h) { skip = /ریشه|ترجم|برگردان|تصريف|تصریف|منابع|مشتق|نفس الجذر/.test(h[1]); continue; }
+ if (skip) continue;
+ const only = line.match(/^[#*]\s*'*\[\[([^\]|]+)\]\]'*\s*(?:\([^)]*\))?\.?\s*$/);
+ if (only) { links.push(only[1]); continue; }
+ const m = line.match(/^#(?![:*])\s*(.+)/);
+ if (!m) continue;
+ let t = m[1];
+ while (/\{\{[^{}]*\}\}/.test(t)) t = t.replace(/\{\{[^{}]*\}\}/g, '');
+ t = unlink(t).replace(/^[\s.،:-]+|[\s]+$/g, '');
+ if (t.length >= 3 && !/^-+$/.test(t)) defs.push(t);
+ }
+ return { defs: defs.slice(0, 5), links };
+}
+
+// meanings from the language's own Wiktionary: the first spelling with an entry (following one pointer hop)
+async function native(code: string, word: string) {
+ const host = `${code}.wiktionary.org`;
+ for (const f of formsIn(code, word)) {
+ const wt = await wikitext(host, f);
+ if (!wt) continue;
+ if (code === 'ur') {
+ const e = urduEntry(wt);
+ if (e.meanings.length) return { defs: e.meanings, origin: e.origin, url: `https://${host}/wiki/${encodeURIComponent(f)}` };
+ continue;
+ }
+ let { defs, links } = definitions(wt, code), title = f;
+ if (!defs.length && links[0]) ({ defs } = definitions(await wikitext(host, (title = links[0])), code));
+ if (defs.length) return { defs, origin: null, url: `https://${host}/wiki/${encodeURIComponent(title)}` };
+ }
+ return null;
+}
+
+// ur.wiktionary: numbered lines under ==معانی==, origin from "(عربی)" on the first line
+export function urduEntry(wt: string) {
+ const m = wt.split(/==\s*معانی\s*==/)[1]?.split(/\n==/)[0] ?? '';
+ const meanings = m.split('\n').map((l) => l.match(/^\s*(?:\d+[.)-]|#)\s*(.+)/)?.[1]).filter(Boolean).map((l) => unlink(l!)).slice(0, 6);
+ const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null;
+ return { meanings, origin };
+}
+
+// en.wiktionary page HTML: per language, IPA, headword transliteration, audio
+export function pronunciations(html: string) {
+ const out: Record = {};
+ const parts = html.split(/]*id="([^"]+)"/);
+ for (let i = 1; i < parts.length; i += 2) {
+ const code = Object.entries(LANGS).find(([, n]) => n === parts[i])?.[0];
+ if (!code) continue;
+ const body = parts[i + 1];
+ const ipa = [...new Set([...body.matchAll(/class="IPA[^"]*"[^>]*>([^<]+)/g)].map((m) => m[1]).filter((x) => /^[/[]/.test(x)))].slice(0, 2);
+ const tr = body.match(new RegExp(`lang="${code}-Latn" class="headword-tr[^"]*"[^>]*>([^<]+)`))?.[1] ?? null;
+ const a = body.match(/"(?:https:)?(\/\/upload\.wikimedia\.org\/[^"]+\.(?:ogg|oga|mp3|wav))"/)?.[1];
+ out[code] = { ipa, tr: tr && plain(tr), audio: a ? 'https:' + a : null };
+ }
+ return out;
+}
+
+// Urdu equivalents of single-word English glosses, from their translation tables
+async function pivot(word: string, glosses: string[]) {
+ const seen = new Set([word.replace(DIAC, '')]); // the word itself, and duplicates differing only in diacritics
+ const words = [...new Set(glosses.flatMap((g) => g.split(/[,;]/)).map((s) => s.trim()).filter((s) => /^[a-z]+$/.test(s)))].slice(0, 3); // lowercase: no proper nouns
+ const found = await Promise.all(words.map(async (gloss) => {
+ const wt = await wikitext('en.wiktionary.org', gloss);
+ const lines = [...wt.matchAll(/^\*:? Urdu: (.*)$/gm)].map((m) => m[1]).join(' ');
+ const urdu = [...lines.matchAll(/\{\{t\+?\|ur\|([^|}]+)/g)].map((m) => m[1].trim())
+ .filter((u) => !seen.has(u.replace(DIAC, '')) && seen.add(u.replace(DIAC, ''))).slice(0, 8);
+ return { gloss, urdu };
+ }));
+ return found.filter((f) => f.urdu.length);
+}
+
+async function fetchWord(word: string) {
+ let title: string | null = null, defs: any = null;
+ for (const f of forms(word)) if ((defs = await get(rest('definition', f)))) { title = f; break; }
+ const [pron, ...own] = await Promise.all([
+ title ? get(rest('html', title), false).then((h) => pronunciations(h ?? '')) : {},
+ ...Object.keys(LANGS).map((c) => native(c, word)),
+ ]) as [Record, ...(Awaited>)[]];
+ const langs = Object.keys(LANGS).map((code, i) => ({
+ code, ...(pron[code] ?? { ipa: [], tr: null, audio: null }),
+ meanings: own[i]?.defs ?? [], origin: own[i]?.origin ?? null, source: own[i]?.url ?? null,
+ senses: (defs?.[code] ?? []).map((e: any) => ({
+ pos: e.partOfSpeech, defs: e.definitions.map((d: any) => plain(d.definition)).filter(Boolean).slice(0, 4),
+ })).filter((s: any) => s.defs.length).slice(0, 3),
+ })).filter((l) => l.meanings.length || l.senses.length || l.ipa.length || l.tr);
+ const glosses = langs.find((l) => l.senses.length)?.senses[0].defs.slice(0, 2) ?? []; // the primary meaning only
+ const hasUrdu = langs.some((l) => l.code === 'ur' && l.meanings.length);
+ return {
+ word, langs,
+ equivalents: hasUrdu ? [] : await pivot(word, glosses),
+ sources: { en: title && `https://en.wiktionary.org/wiki/${encodeURIComponent(title)}` },
+ found: langs.length > 0,
+ };
+}
+
+export async function lookup(word: string) {
+ const hit = await pool.query(`SELECT data FROM dictionary WHERE word = $1 AND fetched_at > now() - interval '30 days'`, [word]);
+ if (hit.rows[0]) return hit.rows[0].data;
+ const data = await fetchWord(word);
+ if (!data.found) return data; // not found (or Wiktionary unreachable): don't cache
+ await pool.query(
+ `INSERT INTO dictionary (word, data) VALUES ($1, $2) ON CONFLICT (word) DO UPDATE SET data = $2, fetched_at = now()`,
+ [word, data],
+ );
+ return data;
+}
diff --git a/api/src/server.ts b/api/src/server.ts
index f8e0ce77..eb72d91b 100644
--- a/api/src/server.ts
+++ b/api/src/server.ts
@@ -2,10 +2,12 @@
// GET /api/poets all poets
// GET /api/page?url=/p238/... poet, category or poem at that URL
// GET /api/search?q=&poet=&page=
+// GET /api/word?w= Wiktionary meanings and pronunciation (sidebar)
// GET /health
import Fastify from 'fastify';
import { pool } from './db.ts';
import { likePatterns, normalise, terms } from './urdu.ts';
+import { lookup } from './dictionary.ts';
const app = Fastify({ logger: { level: process.env.LOG_LEVEL ?? 'info' } });
const PAGE_SIZE = 20;
@@ -124,5 +126,12 @@ app.get<{ Querystring: { q?: string; poet?: string; page?: string } }>('/api/sea
return { total: count.rows[0].n, page, pageSize: PAGE_SIZE, results };
});
+// one word in Arabic script (Urdu, Persian, Arabic), as selected by a reader
+app.get<{ Querystring: { w?: string } }>('/api/word', async (req, reply) => {
+ const w = (req.query.w ?? '').trim();
+ if (!/^[\p{Script=Arabic}\p{M}\u200C]{1,40}$/u.test(w)) return reply.code(400).send({ error: 'one Urdu, Persian or Arabic word' });
+ return lookup(w);
+});
+
const port = Number(process.env.PORT ?? 4100);
await app.listen({ port, host: process.env.HOST ?? '127.0.0.1' });
diff --git a/db/schema.sql b/db/schema.sql
index bbf4c6ed..dc20eff7 100644
--- a/db/schema.sql
+++ b/db/schema.sql
@@ -57,3 +57,10 @@ CREATE TABLE IF NOT EXISTS verses (
text text NOT NULL,
PRIMARY KEY (poem_id, vorder)
);
+
+-- Wiktionary lookups for the reading sidebar (api/src/dictionary.ts), refreshed after 30 days
+CREATE TABLE IF NOT EXISTS dictionary (
+ word text PRIMARY KEY,
+ data jsonb NOT NULL,
+ fetched_at timestamptz NOT NULL DEFAULT now()
+);
diff --git a/web/src/layouts/Base.astro b/web/src/layouts/Base.astro
index 9e7fb578..431b5999 100644
--- a/web/src/layouts/Base.astro
+++ b/web/src/layouts/Base.astro
@@ -48,6 +48,10 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
+