diff --git a/api/src/dictionary.test.ts b/api/src/dictionary.test.ts index 1d4ed537..d15eb835 100644 --- a/api/src/dictionary.test.ts +++ b/api/src/dictionary.test.ts @@ -13,15 +13,25 @@ test('lookup key: spelling variants across Urdu, Persian and Arabic meet', () => assert.notEqual(key('دیوانے'), key('دیوانی'), 'bari ye stays distinct'); }); -test('kaikki entry: glosses, IPA, audio, romanisation, form-of', () => { +test('kaikki entry kept in full: senses with labels and examples, every IPA and recording, form-of', () => { const e = kaikkiEntry({ word: 'قسمت', pos: 'noun', etymology_text: 'From Arabic', - senses: [{ glosses: ['fate, destiny'] }, { glosses: ['division'] }, { links: [] }], - sounds: [{ ipa: '/qɪs.mət̪/' }, { audio: 'x.wav', mp3_url: 'https://upload.wikimedia.org/x.mp3' }, { mp3_url: 'https://evil.example/x.mp3' }], + senses: [{ glosses: ['fate, destiny'], examples: [{ text: 'قسمت کا لکھا', roman: 'qismat kā likhā', english: 'what fate wrote' }] }, + { glosses: ['division'], tags: ['archaic'] }, { links: [] }], + sounds: [{ ipa: '/qɪs.mət̪/', tags: ['Standard'] }, { ipa: '[qɪs.mət̪]' }, { ipa: '/qɪs.mət̪/' }, + { audio: 'x.wav', mp3_url: 'https://upload.wikimedia.org/x.mp3', tags: ['Iran'] }, { ogg_url: 'https://upload.wikimedia.org/y.ogg' }, + { mp3_url: 'https://evil.example/x.mp3' }], forms: [{ form: 'قِسْمَت', tags: ['canonical'] }, { form: 'qismat', tags: ['romanization'] }], + synonyms: [{ word: 'تقدیر' }], derived: [{ word: 'قسمت والا' }, { word: 'قسمت والا' }], + }); + assert.deepEqual(e, { + pos: 'noun', formOf: false, tr: 'qismat', form: 'قِسْمَت', + senses: [{ gloss: 'fate, destiny', tags: [], examples: [{ text: 'قسمت کا لکھا', roman: 'qismat kā likhā', english: 'what fate wrote' }] }, + { gloss: 'division', tags: ['archaic'], examples: [] }], + ipa: [{ ipa: '/qɪs.mət̪/', tags: ['Standard'] }, { ipa: '[qɪs.mət̪]', tags: [] }], + audio: [{ url: 'https://upload.wikimedia.org/x.mp3', tags: ['Iran'] }, { url: 'https://upload.wikimedia.org/y.ogg', tags: [] }], + ety: 'From Arabic', synonyms: ['تقدیر'], derived: ['قسمت والا'], related: [], }); - assert.deepEqual(e, { pos: 'noun', glosses: ['fate, destiny', 'division'], formOf: false, ipa: ['/qɪs.mət̪/'], - audio: 'https://upload.wikimedia.org/x.mp3', tr: 'qismat', form: 'قِسْمَت', ety: 'From Arabic', synonyms: [] }); assert.equal(kaikkiEntry({ word: 'x', pos: 'noun', senses: [{ glosses: ['plural of y'], tags: ['form-of'] }] }).formOf, true); }); @@ -49,9 +59,9 @@ test('dump pages: articles only, entities decoded, redirects skipped', () => { assert.deepEqual([...dumpPages(xml)], [{ title: 'عشق', text: 'a & b ' }]); }); -test('etymology cut never leaves half a surrogate pair', () => { - const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'a'.repeat(299) + '𑀭𑀭' }).ety!; - assert.ok(ety.isWellFormed()); +test('etymology kept in full and well formed', () => { + const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'a'.repeat(400) + '𑀭\ud804' }).ety!; + assert.ok(ety.isWellFormed() && ety.length > 400); }); test('ur.wiktionary: English entries give Urdu words; Urdu entries fall back to # lines or prose', async () => { diff --git a/api/src/dictionary.ts b/api/src/dictionary.ts index 7dad871f..986c121a 100644 --- a/api/src/dictionary.ts +++ b/api/src/dictionary.ts @@ -1,5 +1,7 @@ // Word dictionary for the reading sidebar: the full Wiktionary data for Urdu, Persian and Arabic, in PostgreSQL. -// source 'en': en.wiktionary entries (English meanings, IPA, transliteration, audio, etymology, synonyms), +// source 'en': en.wiktionary entries in full (all senses with usage examples, every IPA with dialect labels, every +// recording, transliteration, vowelled form, etymology, synonyms, derived and related words; +// not inflection tables), // from the kaikki.org Wiktextract extracts (weekly) // source 'own': each language's own Wiktionary (ur., fa., ar.wiktionary): meanings in that language, // from the Wikimedia dumps (twice a month) plus recent changes (daily) @@ -37,24 +39,32 @@ const upload = (u?: string) => (u && u.startsWith('https://upload.wikimedia.org/ export function kaikkiEntry(d: any) { const senses = (d.senses ?? []).filter((s: any) => s.glosses?.length); const sounds = d.sounds ?? []; + const words = (list: any[] | undefined) => [...new Set((list ?? []).map((x: any) => x.word).filter(Boolean))] as string[]; + const ipa = new Map(); + for (const s of sounds) if (s.ipa && !ipa.has(s.ipa)) ipa.set(s.ipa, s.tags ?? []); return { pos: d.pos as string, - glosses: senses.map((s: any) => s.glosses.join('; ')).slice(0, 6) as string[], formOf: senses.length > 0 && senses.every((s: any) => s.form_of || s.tags?.includes('form-of')), - ipa: [...new Set(sounds.map((s: any) => s.ipa).filter(Boolean))].slice(0, 2) as string[], - audio: sounds.map((s: any) => upload(s.mp3_url) ?? upload(s.ogg_url)).find(Boolean) ?? null, tr: d.forms?.find((f: any) => f.tags?.includes('romanization'))?.form ?? null, form: d.forms?.find((f: any) => f.tags?.includes('canonical'))?.form ?? null, // with short vowels: مُلْک + // every sense, with its labels (archaic, figurative …) and usage examples + senses: senses.map((s: any) => ({ + gloss: s.glosses.join('; ') as string, + tags: (s.tags ?? []).filter((t: string) => t !== 'form-of') as string[], + examples: (s.examples ?? []).filter((e: any) => e.text).map((e: any) => ({ text: e.text, roman: e.roman ?? null, english: e.english ?? e.translation ?? null })), + })), + // every pronunciation with its dialect labels (Iran, Classical Persian …) and every recording + ipa: [...ipa].map(([v, tags]) => ({ ipa: v, tags })), + audio: sounds.map((s: any) => ({ url: upload(s.mp3_url) ?? upload(s.ogg_url), tags: s.tags ?? [] })).filter((a: any) => a.url), ety: d.etymology_text ? etymology(String(d.etymology_text)) : null, - synonyms: (d.synonyms ?? []).map((s: any) => s.word).filter(Boolean).slice(0, 8) as string[], + synonyms: words(d.synonyms), derived: words(d.derived), related: words(d.related), }; } -// etymology text without Wiktionary's "Etymology tree …" diagram summary; cut at 300 characters, never -// leaving half a surrogate pair +// etymology text without Wiktionary's "Etymology tree …" diagram summary (no half surrogate pairs) const etymology = (t: string) => (t.startsWith('Etymology tree') ? t.slice(Math.max(0, t.search(/\b(Borrowed|Inherited|From|Learned|Derived|Semi-learned|Calque|Compound)\b/))) : t) - .slice(0, 300).toWellFormed(); + .toWellFormed(); // English glosses of an Urdu entry as pivot keys: "fate, destiny" -> fate, destiny (short, lowercase) export const glossKeys = (glosses: string[]) => [...new Set(glosses.flatMap((g) => g.replace(/\([^)]*\)/g, '').split(/[,;]/)) @@ -73,7 +83,7 @@ export function urduEntry(wt: string) { } const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null; const english = wt.match(/^\s*انگریزی\s*:\s*([A-Za-z].+)$/m)?.[1].trim() ?? null; // "== تراجم ==" pages - return { defs: defs.slice(0, 6), origin, links: [] as string[], ...(english && { english }) }; + return { defs: defs.slice(0, 20), origin, links: [] as string[], ...(english && { english }) }; } // ur.wiktionary English entries ("north": "# [[شمالی]]۔"): the Urdu words, for the pivot through English @@ -105,7 +115,7 @@ export function definitions(wt: string, code: string) { const t = unlink(stripTemplates(m[1])).replace(/^[\s.،:-]+|\s+$/g, ''); if (t.length >= 3 && !/^-+$/.test(t)) defs.push(t); } - return { defs: defs.slice(0, 6), origin: null as string | null, links: links.slice(0, 5) }; + return { defs: defs.slice(0, 20), origin: null as string | null, links: links.slice(0, 5) }; } export const ownEntry = (code: string, wt: string) => (code === 'ur' ? urduEntry(wt) : definitions(wt, code)); @@ -174,9 +184,9 @@ async function importKaikki(lang: string) { for await (const line of createInterface({ input: Readable.fromWeb(res.body as any), crlfDelay: Infinity })) { if (!line) continue; const d = JSON.parse(line), e = kaikkiEntry(d); - if (!e.glosses.length && !e.ipa.length) continue; + if (!e.senses.length && !e.ipa.length) continue; await add(d.word, e); - if (lang === 'ur' && !e.formOf) for (const g of glossKeys(e.glosses.slice(0, 3))) glossRows.push([g, d.word]); + if (lang === 'ur' && !e.formOf) for (const g of glossKeys(e.senses.slice(0, 3).map((s) => s.gloss))) glossRows.push([g, d.word]); } await insertGlosses(client, 'en', glossRows); }); @@ -291,16 +301,18 @@ export async function lookup(word: string) { for (const r of use) { const id = r.data.tr ?? r.data.form ?? r.title; let g = readings.find((x) => x.id === id); - if (!g) readings.push((g = { id, form: r.data.form ?? r.title, tr: r.data.tr, ipa: r.data.ipa, audio: r.data.audio, senses: [] })); - if (!g.ipa.length) g.ipa = r.data.ipa; - g.audio ??= r.data.audio; - if (r.data.glosses.length && g.senses.length < 3) g.senses.push({ pos: r.data.pos, defs: r.data.glosses.slice(0, 4) }); + if (!g) readings.push((g = { id, form: r.data.form ?? r.title, tr: r.data.tr, ipa: [], audio: [], senses: [] })); + for (const x of r.data.ipa) if (!g.ipa.some((y: any) => y.ipa === x.ipa)) g.ipa.push(x); + for (const x of r.data.audio) if (!g.audio.some((y: any) => y.url === x.url)) g.audio.push(x); + if (r.data.senses.length) g.senses.push({ pos: r.data.pos, defs: r.data.senses }); } return { - code, readings: readings.slice(0, 5).map(({ id, ...x }) => x), - ety: use.map((r) => r.data.ety).find(Boolean) ?? null, - synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))].slice(0, 8), - meanings: own.flatMap((r) => r.data.defs).slice(0, 6), origin: own.find((r) => r.data.origin)?.data.origin ?? null, + code, readings: readings.map(({ id, ...x }) => x), + etymologies: [...new Set(use.map((r) => r.data.ety).filter(Boolean))], + synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))], + derived: [...new Set(use.flatMap((r) => r.data.derived))], + related: [...new Set(use.flatMap((r) => r.data.related))], + meanings: own.flatMap((r) => r.data.defs), origin: own.find((r) => r.data.origin)?.data.origin ?? null, source: own[0] ? page(code, 'own', own[0].title) : null, // ur.wiktionary's English translation lines ("انگریزی : …") translation: rows.filter((r) => r.lang === code && r.source === 'own' && r.data.english).map((r) => r.data.english), @@ -311,7 +323,7 @@ export async function lookup(word: string) { // has no Urdu-language meaning, and for the Persian and Arabic entries when the word is not in Urdu at all const inUrdu = langs.some((l) => l.code === 'ur'); await Promise.all(langs.map(async (l: any) => { - l.urdu = (l.code === 'ur' ? !l.meanings.length : !inUrdu) ? await viaEnglish(l.readings[0]?.senses[0]?.defs.slice(0, 2) ?? [], k) : []; + l.urdu = (l.code === 'ur' ? !l.meanings.length : !inUrdu) ? await viaEnglish(l.readings[0]?.senses[0]?.defs.slice(0, 2).map((d: any) => d.gloss) ?? [], k) : []; })); const found = langs.length > 0; // always: the same consonants with other long vowels (دل: دال، دول، دیل); not found: also stems and near spellings diff --git a/web/src/layouts/Base.astro b/web/src/layouts/Base.astro index 2a748e33..6c4bb5f3 100644 --- a/web/src/layouts/Base.astro +++ b/web/src/layouts/Base.astro @@ -136,18 +136,18 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک dictBody.replaceChildren(el('h2', null, w), el('p', 'muted', 'تلاش جاری ہے…')); let d = null; // v: bump when the response format changes (responses are browser-cached for a day) - try { const r = await fetch('/api/word?v=5&w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {} + try { const r = await fetch('/api/word?v=6&w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {} if (w !== current) return; // a newer selection won - const words = (title, items, into) => { // clickable word buttons - if (!items.length) return; + const buttons = (items) => { // clickable words const box = el('div', 'similar'); for (const s of items) { const b = el('button', null, s.title, { type: 'button', title: s.langs.map((c) => LANG[c]).join('، ') }); b.onclick = () => showWord(s.title); box.append(b); } - into.push(el('h3', null, title), box); + return box; }; + const words = (title, items, into) => { if (items.length) into.push(el('h3', null, title), buttons(items)); }; const list = (items, cls, attrs) => { const ol = el('ol', cls, null, attrs); items.forEach((x) => ol.append(el('li', null, x))); return ol; }; const out = [el('h2', null, w)]; if (!d || !d.found) { @@ -161,19 +161,50 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک for (const l of d.langs) { const sec = el('section', 'dict-lang'); sec.append(el('h3', null, LANG[l.code])); - // a pronunciation line: vowelled form, transliteration, IPA, audio + // collapsible extra (more meanings, examples, long etymologies, word lists) + const more = (label, ...children) => { + const d = el('details', 'more'); + d.append(el('summary', null, label), ...children); + return d; + }; + const tagged = (tags) => (tags?.length ? ` (${tags.join(', ')})` : ''); + // a pronunciation line: vowelled form, transliteration, every IPA (with dialect labels), every recording const pron = (r) => { const pr = el('p', 'pron'); if (r.form) pr.append(el('span', 'form', r.form), ' '); if (r.tr) pr.append(el('bdi', 'en', r.tr)); - r.ipa.forEach((i) => pr.append(' ', el('bdi', 'ipa', i))); - if (r.audio && r.audio.startsWith('https://upload.wikimedia.org/')) { - const b = el('button', 'play', '🔊', { type: 'button', 'aria-label': 'تلفظ سنیں' }); - b.onclick = () => new Audio(r.audio).play(); + r.ipa.forEach((i) => pr.append(' ', el('bdi', 'ipa', i.ipa + tagged(i.tags), { title: i.tags.join(', ') }))); + r.audio.forEach((a) => { + if (!a.url.startsWith('https://upload.wikimedia.org/')) return; + const b = el('button', 'play', '🔊', { type: 'button', 'aria-label': 'تلفظ سنیں', title: a.tags.join(', ') }); + b.onclick = () => new Audio(a.url).play(); pr.append(' ', b); - } + }); return pr; }; + // one meaning: gloss with labels; usage examples (text, romanisation, translation) folded + const sense = (x) => { + const li = el('li', null, x.gloss + tagged(x.tags)); + if (x.examples.length) { + const ex = x.examples.map((e) => { + const p = el('p', 'example'); + p.append(el('span', 'ex-text', e.text, { dir: 'auto' })); + if (e.roman) p.append(el('br'), el('i', null, e.roman)); + if (e.english) p.append(el('br'), e.english); + return p; + }); + li.append(more(`مثالیں (${x.examples.length})`, ...ex)); + } + return li; + }; + const senses = (defs) => { + const ol = el('ol', 'en', null, { dir: 'ltr', lang: 'en' }); + defs.slice(0, 4).forEach((x) => ol.append(sense(x))); + if (defs.length <= 4) return [ol]; + const rest = el('ol', 'en', null, { dir: 'ltr', lang: 'en', start: '5' }); + defs.slice(4).forEach((x) => rest.append(sense(x))); + return [ol, more(`مزید معانی (${defs.length - 4})`, rest)]; + }; const urdu = () => { if (!l.urdu?.length) return; sec.append(el('p', 'pos', 'اردو میں · انگریزی ویکی لغت کے واسطے سے')); @@ -189,7 +220,7 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک for (const r of l.readings) { const box = el('div', l.readings.length > 1 ? 'reading' : null); box.append(pron(r)); - for (const s of r.senses) box.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos.charAt(0).toUpperCase() + s.pos.slice(1)] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' })); + for (const s of r.senses) box.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos.charAt(0).toUpperCase() + s.pos.slice(1)] ?? s.pos}`), ...senses(s.defs)); sec.append(box); } if (l.translation.length) sec.append(el('p', 'pos', POS.Translation), list(l.translation, 'en', { dir: 'ltr', lang: 'en' })); @@ -199,11 +230,20 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک const label = el('p', 'pos'); label.append(`معانی: ${LANG[l.code]} ویکی لغت` + (l.origin ? ` · اصل: ${l.origin}` : '') + ' · '); if (l.source) label.append(el('a', null, 'ماخذ', { href: l.source, target: '_blank', rel: 'noopener' })); - sec.append(label, list(l.meanings, null, { lang: l.code })); + const own = list(l.meanings.slice(0, 6), null, { lang: l.code }); + sec.append(label, own); + if (l.meanings.length > 6) sec.append(more(`مزید معانی (${l.meanings.length - 6})`, list(l.meanings.slice(6), null, { lang: l.code, start: '7' }))); } if (l.code === 'ur') { english(); urdu(); } - if (l.synonyms.length) sec.append(el('p', 'pos', 'مترادفات'), el('p', null, l.synonyms.join('، '))); - if (l.ety) sec.append(el('p', 'pos', 'اشتقاق'), el('p', 'en ety', l.ety, { dir: 'ltr', lang: 'en' })); + for (const [label, items] of [['مترادفات', l.synonyms], ['مشتقات', l.derived], ['متعلقہ الفاظ', l.related]]) { + if (!items.length) continue; + const box = buttons(items.map((t) => ({ title: t, langs: [l.code] }))); + sec.append(...(items.length > 8 ? [more(`${label} (${items.length})`, box)] : [el('p', 'pos', label), box])); + } + for (const ety of l.etymologies) { + const text = el('p', 'en ety', ety, { dir: 'ltr', lang: 'en' }); + sec.append(...(ety.length > 220 ? [more('اشتقاق', text)] : [el('p', 'pos', 'اشتقاق'), text])); + } if (l.en) { const p = el('p', 'pos'); p.append(el('a', null, 'انگریزی ویکی لغت', { href: l.en, target: '_blank', rel: 'noopener' })); diff --git a/web/src/styles/global.css b/web/src/styles/global.css index 3e2158e9..98a07d74 100644 --- a/web/src/styles/global.css +++ b/web/src/styles/global.css @@ -146,6 +146,10 @@ h1 + .muted { text-align: center; margin-top: 0; } .dict .pos { font-size: .8rem; color: var(--muted); margin: 8px 0 0; } .dict .note { font-size: .8rem; margin: 0; } .dict .via { margin: 2px 0; } +.dict details.more { margin: 4px 0; } +.dict details.more summary { cursor: pointer; font-size: .8rem; color: var(--brand); } +.dict .example { font-size: .8rem; color: var(--muted); margin: 4px 0; border-left: 2px solid var(--gold-light); padding-left: 8px; } +.dict .example .ex-text { font-family: 'Noto Naskh Arabic', serif; font-size: .95rem; color: var(--ink); } .dict .reading { border-inline-start: 2px solid var(--gold-light); padding-inline-start: 10px; margin: 8px 0; } .dict .form { font-size: 1.15rem; color: var(--lapis); } .dict .via .en { font-size: .78rem; }