Word dictionary in full: all senses with usage examples, every IPA with dialect labels, every recording, full etymologies, synonyms, derived and related words
Panel folds the extras (more meanings, examples, long etymologies, long word lists). 175 MB in PostgreSQL. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
19b68425f6
commit
3201a1c478
@ -13,15 +13,25 @@ test('lookup key: spelling variants across Urdu, Persian and Arabic meet', () =>
|
||||
assert.notEqual(key('دیوانے'), key('دیوانی'), 'bari ye stays distinct');
|
||||
});
|
||||
|
||||
test('kaikki entry: glosses, IPA, audio, romanisation, form-of', () => {
|
||||
test('kaikki entry kept in full: senses with labels and examples, every IPA and recording, form-of', () => {
|
||||
const e = kaikkiEntry({
|
||||
word: 'قسمت', pos: 'noun', etymology_text: 'From Arabic',
|
||||
senses: [{ glosses: ['fate, destiny'] }, { glosses: ['division'] }, { links: [] }],
|
||||
sounds: [{ ipa: '/qɪs.mət̪/' }, { audio: 'x.wav', mp3_url: 'https://upload.wikimedia.org/x.mp3' }, { mp3_url: 'https://evil.example/x.mp3' }],
|
||||
senses: [{ glosses: ['fate, destiny'], examples: [{ text: 'قسمت کا لکھا', roman: 'qismat kā likhā', english: 'what fate wrote' }] },
|
||||
{ glosses: ['division'], tags: ['archaic'] }, { links: [] }],
|
||||
sounds: [{ ipa: '/qɪs.mət̪/', tags: ['Standard'] }, { ipa: '[qɪs.mət̪]' }, { ipa: '/qɪs.mət̪/' },
|
||||
{ audio: 'x.wav', mp3_url: 'https://upload.wikimedia.org/x.mp3', tags: ['Iran'] }, { ogg_url: 'https://upload.wikimedia.org/y.ogg' },
|
||||
{ mp3_url: 'https://evil.example/x.mp3' }],
|
||||
forms: [{ form: 'قِسْمَت', tags: ['canonical'] }, { form: 'qismat', tags: ['romanization'] }],
|
||||
synonyms: [{ word: 'تقدیر' }], derived: [{ word: 'قسمت والا' }, { word: 'قسمت والا' }],
|
||||
});
|
||||
assert.deepEqual(e, {
|
||||
pos: 'noun', formOf: false, tr: 'qismat', form: 'قِسْمَت',
|
||||
senses: [{ gloss: 'fate, destiny', tags: [], examples: [{ text: 'قسمت کا لکھا', roman: 'qismat kā likhā', english: 'what fate wrote' }] },
|
||||
{ gloss: 'division', tags: ['archaic'], examples: [] }],
|
||||
ipa: [{ ipa: '/qɪs.mət̪/', tags: ['Standard'] }, { ipa: '[qɪs.mət̪]', tags: [] }],
|
||||
audio: [{ url: 'https://upload.wikimedia.org/x.mp3', tags: ['Iran'] }, { url: 'https://upload.wikimedia.org/y.ogg', tags: [] }],
|
||||
ety: 'From Arabic', synonyms: ['تقدیر'], derived: ['قسمت والا'], related: [],
|
||||
});
|
||||
assert.deepEqual(e, { pos: 'noun', glosses: ['fate, destiny', 'division'], formOf: false, ipa: ['/qɪs.mət̪/'],
|
||||
audio: 'https://upload.wikimedia.org/x.mp3', tr: 'qismat', form: 'قِسْمَت', ety: 'From Arabic', synonyms: [] });
|
||||
assert.equal(kaikkiEntry({ word: 'x', pos: 'noun', senses: [{ glosses: ['plural of y'], tags: ['form-of'] }] }).formOf, true);
|
||||
});
|
||||
|
||||
@ -49,9 +59,9 @@ test('dump pages: articles only, entities decoded, redirects skipped', () => {
|
||||
assert.deepEqual([...dumpPages(xml)], [{ title: 'عشق', text: 'a & b <x>' }]);
|
||||
});
|
||||
|
||||
test('etymology cut never leaves half a surrogate pair', () => {
|
||||
const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'a'.repeat(299) + '𑀭𑀭' }).ety!;
|
||||
assert.ok(ety.isWellFormed());
|
||||
test('etymology kept in full and well formed', () => {
|
||||
const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'a'.repeat(400) + '𑀭\ud804' }).ety!;
|
||||
assert.ok(ety.isWellFormed() && ety.length > 400);
|
||||
});
|
||||
|
||||
test('ur.wiktionary: English entries give Urdu words; Urdu entries fall back to # lines or prose', async () => {
|
||||
|
||||
@ -1,5 +1,7 @@
|
||||
// Word dictionary for the reading sidebar: the full Wiktionary data for Urdu, Persian and Arabic, in PostgreSQL.
|
||||
// source 'en': en.wiktionary entries (English meanings, IPA, transliteration, audio, etymology, synonyms),
|
||||
// source 'en': en.wiktionary entries in full (all senses with usage examples, every IPA with dialect labels, every
|
||||
// recording, transliteration, vowelled form, etymology, synonyms, derived and related words;
|
||||
// not inflection tables),
|
||||
// from the kaikki.org Wiktextract extracts (weekly)
|
||||
// source 'own': each language's own Wiktionary (ur., fa., ar.wiktionary): meanings in that language,
|
||||
// from the Wikimedia dumps (twice a month) plus recent changes (daily)
|
||||
@ -37,24 +39,32 @@ const upload = (u?: string) => (u && u.startsWith('https://upload.wikimedia.org/
|
||||
export function kaikkiEntry(d: any) {
|
||||
const senses = (d.senses ?? []).filter((s: any) => s.glosses?.length);
|
||||
const sounds = d.sounds ?? [];
|
||||
const words = (list: any[] | undefined) => [...new Set((list ?? []).map((x: any) => x.word).filter(Boolean))] as string[];
|
||||
const ipa = new Map<string, string[]>();
|
||||
for (const s of sounds) if (s.ipa && !ipa.has(s.ipa)) ipa.set(s.ipa, s.tags ?? []);
|
||||
return {
|
||||
pos: d.pos as string,
|
||||
glosses: senses.map((s: any) => s.glosses.join('; ')).slice(0, 6) as string[],
|
||||
formOf: senses.length > 0 && senses.every((s: any) => s.form_of || s.tags?.includes('form-of')),
|
||||
ipa: [...new Set(sounds.map((s: any) => s.ipa).filter(Boolean))].slice(0, 2) as string[],
|
||||
audio: sounds.map((s: any) => upload(s.mp3_url) ?? upload(s.ogg_url)).find(Boolean) ?? null,
|
||||
tr: d.forms?.find((f: any) => f.tags?.includes('romanization'))?.form ?? null,
|
||||
form: d.forms?.find((f: any) => f.tags?.includes('canonical'))?.form ?? null, // with short vowels: مُلْک
|
||||
// every sense, with its labels (archaic, figurative …) and usage examples
|
||||
senses: senses.map((s: any) => ({
|
||||
gloss: s.glosses.join('; ') as string,
|
||||
tags: (s.tags ?? []).filter((t: string) => t !== 'form-of') as string[],
|
||||
examples: (s.examples ?? []).filter((e: any) => e.text).map((e: any) => ({ text: e.text, roman: e.roman ?? null, english: e.english ?? e.translation ?? null })),
|
||||
})),
|
||||
// every pronunciation with its dialect labels (Iran, Classical Persian …) and every recording
|
||||
ipa: [...ipa].map(([v, tags]) => ({ ipa: v, tags })),
|
||||
audio: sounds.map((s: any) => ({ url: upload(s.mp3_url) ?? upload(s.ogg_url), tags: s.tags ?? [] })).filter((a: any) => a.url),
|
||||
ety: d.etymology_text ? etymology(String(d.etymology_text)) : null,
|
||||
synonyms: (d.synonyms ?? []).map((s: any) => s.word).filter(Boolean).slice(0, 8) as string[],
|
||||
synonyms: words(d.synonyms), derived: words(d.derived), related: words(d.related),
|
||||
};
|
||||
}
|
||||
|
||||
// etymology text without Wiktionary's "Etymology tree …" diagram summary; cut at 300 characters, never
|
||||
// leaving half a surrogate pair
|
||||
// etymology text without Wiktionary's "Etymology tree …" diagram summary (no half surrogate pairs)
|
||||
const etymology = (t: string) =>
|
||||
(t.startsWith('Etymology tree') ? t.slice(Math.max(0, t.search(/\b(Borrowed|Inherited|From|Learned|Derived|Semi-learned|Calque|Compound)\b/))) : t)
|
||||
.slice(0, 300).toWellFormed();
|
||||
.toWellFormed();
|
||||
|
||||
// English glosses of an Urdu entry as pivot keys: "fate, destiny" -> fate, destiny (short, lowercase)
|
||||
export const glossKeys = (glosses: string[]) => [...new Set(glosses.flatMap((g) => g.replace(/\([^)]*\)/g, '').split(/[,;]/))
|
||||
@ -73,7 +83,7 @@ export function urduEntry(wt: string) {
|
||||
}
|
||||
const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null;
|
||||
const english = wt.match(/^\s*انگریزی\s*:\s*([A-Za-z].+)$/m)?.[1].trim() ?? null; // "== تراجم ==" pages
|
||||
return { defs: defs.slice(0, 6), origin, links: [] as string[], ...(english && { english }) };
|
||||
return { defs: defs.slice(0, 20), origin, links: [] as string[], ...(english && { english }) };
|
||||
}
|
||||
|
||||
// ur.wiktionary English entries ("north": "# [[شمالی]]۔"): the Urdu words, for the pivot through English
|
||||
@ -105,7 +115,7 @@ export function definitions(wt: string, code: string) {
|
||||
const t = unlink(stripTemplates(m[1])).replace(/^[\s.،:-]+|\s+$/g, '');
|
||||
if (t.length >= 3 && !/^-+$/.test(t)) defs.push(t);
|
||||
}
|
||||
return { defs: defs.slice(0, 6), origin: null as string | null, links: links.slice(0, 5) };
|
||||
return { defs: defs.slice(0, 20), origin: null as string | null, links: links.slice(0, 5) };
|
||||
}
|
||||
|
||||
export const ownEntry = (code: string, wt: string) => (code === 'ur' ? urduEntry(wt) : definitions(wt, code));
|
||||
@ -174,9 +184,9 @@ async function importKaikki(lang: string) {
|
||||
for await (const line of createInterface({ input: Readable.fromWeb(res.body as any), crlfDelay: Infinity })) {
|
||||
if (!line) continue;
|
||||
const d = JSON.parse(line), e = kaikkiEntry(d);
|
||||
if (!e.glosses.length && !e.ipa.length) continue;
|
||||
if (!e.senses.length && !e.ipa.length) continue;
|
||||
await add(d.word, e);
|
||||
if (lang === 'ur' && !e.formOf) for (const g of glossKeys(e.glosses.slice(0, 3))) glossRows.push([g, d.word]);
|
||||
if (lang === 'ur' && !e.formOf) for (const g of glossKeys(e.senses.slice(0, 3).map((s) => s.gloss))) glossRows.push([g, d.word]);
|
||||
}
|
||||
await insertGlosses(client, 'en', glossRows);
|
||||
});
|
||||
@ -291,16 +301,18 @@ export async function lookup(word: string) {
|
||||
for (const r of use) {
|
||||
const id = r.data.tr ?? r.data.form ?? r.title;
|
||||
let g = readings.find((x) => x.id === id);
|
||||
if (!g) readings.push((g = { id, form: r.data.form ?? r.title, tr: r.data.tr, ipa: r.data.ipa, audio: r.data.audio, senses: [] }));
|
||||
if (!g.ipa.length) g.ipa = r.data.ipa;
|
||||
g.audio ??= r.data.audio;
|
||||
if (r.data.glosses.length && g.senses.length < 3) g.senses.push({ pos: r.data.pos, defs: r.data.glosses.slice(0, 4) });
|
||||
if (!g) readings.push((g = { id, form: r.data.form ?? r.title, tr: r.data.tr, ipa: [], audio: [], senses: [] }));
|
||||
for (const x of r.data.ipa) if (!g.ipa.some((y: any) => y.ipa === x.ipa)) g.ipa.push(x);
|
||||
for (const x of r.data.audio) if (!g.audio.some((y: any) => y.url === x.url)) g.audio.push(x);
|
||||
if (r.data.senses.length) g.senses.push({ pos: r.data.pos, defs: r.data.senses });
|
||||
}
|
||||
return {
|
||||
code, readings: readings.slice(0, 5).map(({ id, ...x }) => x),
|
||||
ety: use.map((r) => r.data.ety).find(Boolean) ?? null,
|
||||
synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))].slice(0, 8),
|
||||
meanings: own.flatMap((r) => r.data.defs).slice(0, 6), origin: own.find((r) => r.data.origin)?.data.origin ?? null,
|
||||
code, readings: readings.map(({ id, ...x }) => x),
|
||||
etymologies: [...new Set(use.map((r) => r.data.ety).filter(Boolean))],
|
||||
synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))],
|
||||
derived: [...new Set(use.flatMap((r) => r.data.derived))],
|
||||
related: [...new Set(use.flatMap((r) => r.data.related))],
|
||||
meanings: own.flatMap((r) => r.data.defs), origin: own.find((r) => r.data.origin)?.data.origin ?? null,
|
||||
source: own[0] ? page(code, 'own', own[0].title) : null,
|
||||
// ur.wiktionary's English translation lines ("انگریزی : …")
|
||||
translation: rows.filter((r) => r.lang === code && r.source === 'own' && r.data.english).map((r) => r.data.english),
|
||||
@ -311,7 +323,7 @@ export async function lookup(word: string) {
|
||||
// has no Urdu-language meaning, and for the Persian and Arabic entries when the word is not in Urdu at all
|
||||
const inUrdu = langs.some((l) => l.code === 'ur');
|
||||
await Promise.all(langs.map(async (l: any) => {
|
||||
l.urdu = (l.code === 'ur' ? !l.meanings.length : !inUrdu) ? await viaEnglish(l.readings[0]?.senses[0]?.defs.slice(0, 2) ?? [], k) : [];
|
||||
l.urdu = (l.code === 'ur' ? !l.meanings.length : !inUrdu) ? await viaEnglish(l.readings[0]?.senses[0]?.defs.slice(0, 2).map((d: any) => d.gloss) ?? [], k) : [];
|
||||
}));
|
||||
const found = langs.length > 0;
|
||||
// always: the same consonants with other long vowels (دل: دال، دول، دیل); not found: also stems and near spellings
|
||||
|
||||
@ -136,18 +136,18 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
|
||||
dictBody.replaceChildren(el('h2', null, w), el('p', 'muted', 'تلاش جاری ہے…'));
|
||||
let d = null;
|
||||
// v: bump when the response format changes (responses are browser-cached for a day)
|
||||
try { const r = await fetch('/api/word?v=5&w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {}
|
||||
try { const r = await fetch('/api/word?v=6&w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {}
|
||||
if (w !== current) return; // a newer selection won
|
||||
const words = (title, items, into) => { // clickable word buttons
|
||||
if (!items.length) return;
|
||||
const buttons = (items) => { // clickable words
|
||||
const box = el('div', 'similar');
|
||||
for (const s of items) {
|
||||
const b = el('button', null, s.title, { type: 'button', title: s.langs.map((c) => LANG[c]).join('، ') });
|
||||
b.onclick = () => showWord(s.title);
|
||||
box.append(b);
|
||||
}
|
||||
into.push(el('h3', null, title), box);
|
||||
return box;
|
||||
};
|
||||
const words = (title, items, into) => { if (items.length) into.push(el('h3', null, title), buttons(items)); };
|
||||
const list = (items, cls, attrs) => { const ol = el('ol', cls, null, attrs); items.forEach((x) => ol.append(el('li', null, x))); return ol; };
|
||||
const out = [el('h2', null, w)];
|
||||
if (!d || !d.found) {
|
||||
@ -161,19 +161,50 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
|
||||
for (const l of d.langs) {
|
||||
const sec = el('section', 'dict-lang');
|
||||
sec.append(el('h3', null, LANG[l.code]));
|
||||
// a pronunciation line: vowelled form, transliteration, IPA, audio
|
||||
// collapsible extra (more meanings, examples, long etymologies, word lists)
|
||||
const more = (label, ...children) => {
|
||||
const d = el('details', 'more');
|
||||
d.append(el('summary', null, label), ...children);
|
||||
return d;
|
||||
};
|
||||
const tagged = (tags) => (tags?.length ? ` (${tags.join(', ')})` : '');
|
||||
// a pronunciation line: vowelled form, transliteration, every IPA (with dialect labels), every recording
|
||||
const pron = (r) => {
|
||||
const pr = el('p', 'pron');
|
||||
if (r.form) pr.append(el('span', 'form', r.form), ' ');
|
||||
if (r.tr) pr.append(el('bdi', 'en', r.tr));
|
||||
r.ipa.forEach((i) => pr.append(' ', el('bdi', 'ipa', i)));
|
||||
if (r.audio && r.audio.startsWith('https://upload.wikimedia.org/')) {
|
||||
const b = el('button', 'play', '🔊', { type: 'button', 'aria-label': 'تلفظ سنیں' });
|
||||
b.onclick = () => new Audio(r.audio).play();
|
||||
r.ipa.forEach((i) => pr.append(' ', el('bdi', 'ipa', i.ipa + tagged(i.tags), { title: i.tags.join(', ') })));
|
||||
r.audio.forEach((a) => {
|
||||
if (!a.url.startsWith('https://upload.wikimedia.org/')) return;
|
||||
const b = el('button', 'play', '🔊', { type: 'button', 'aria-label': 'تلفظ سنیں', title: a.tags.join(', ') });
|
||||
b.onclick = () => new Audio(a.url).play();
|
||||
pr.append(' ', b);
|
||||
}
|
||||
});
|
||||
return pr;
|
||||
};
|
||||
// one meaning: gloss with labels; usage examples (text, romanisation, translation) folded
|
||||
const sense = (x) => {
|
||||
const li = el('li', null, x.gloss + tagged(x.tags));
|
||||
if (x.examples.length) {
|
||||
const ex = x.examples.map((e) => {
|
||||
const p = el('p', 'example');
|
||||
p.append(el('span', 'ex-text', e.text, { dir: 'auto' }));
|
||||
if (e.roman) p.append(el('br'), el('i', null, e.roman));
|
||||
if (e.english) p.append(el('br'), e.english);
|
||||
return p;
|
||||
});
|
||||
li.append(more(`مثالیں (${x.examples.length})`, ...ex));
|
||||
}
|
||||
return li;
|
||||
};
|
||||
const senses = (defs) => {
|
||||
const ol = el('ol', 'en', null, { dir: 'ltr', lang: 'en' });
|
||||
defs.slice(0, 4).forEach((x) => ol.append(sense(x)));
|
||||
if (defs.length <= 4) return [ol];
|
||||
const rest = el('ol', 'en', null, { dir: 'ltr', lang: 'en', start: '5' });
|
||||
defs.slice(4).forEach((x) => rest.append(sense(x)));
|
||||
return [ol, more(`مزید معانی (${defs.length - 4})`, rest)];
|
||||
};
|
||||
const urdu = () => {
|
||||
if (!l.urdu?.length) return;
|
||||
sec.append(el('p', 'pos', 'اردو میں · انگریزی ویکی لغت کے واسطے سے'));
|
||||
@ -189,7 +220,7 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
|
||||
for (const r of l.readings) {
|
||||
const box = el('div', l.readings.length > 1 ? 'reading' : null);
|
||||
box.append(pron(r));
|
||||
for (const s of r.senses) box.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos.charAt(0).toUpperCase() + s.pos.slice(1)] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' }));
|
||||
for (const s of r.senses) box.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos.charAt(0).toUpperCase() + s.pos.slice(1)] ?? s.pos}`), ...senses(s.defs));
|
||||
sec.append(box);
|
||||
}
|
||||
if (l.translation.length) sec.append(el('p', 'pos', POS.Translation), list(l.translation, 'en', { dir: 'ltr', lang: 'en' }));
|
||||
@ -199,11 +230,20 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
|
||||
const label = el('p', 'pos');
|
||||
label.append(`معانی: ${LANG[l.code]} ویکی لغت` + (l.origin ? ` · اصل: ${l.origin}` : '') + ' · ');
|
||||
if (l.source) label.append(el('a', null, 'ماخذ', { href: l.source, target: '_blank', rel: 'noopener' }));
|
||||
sec.append(label, list(l.meanings, null, { lang: l.code }));
|
||||
const own = list(l.meanings.slice(0, 6), null, { lang: l.code });
|
||||
sec.append(label, own);
|
||||
if (l.meanings.length > 6) sec.append(more(`مزید معانی (${l.meanings.length - 6})`, list(l.meanings.slice(6), null, { lang: l.code, start: '7' })));
|
||||
}
|
||||
if (l.code === 'ur') { english(); urdu(); }
|
||||
if (l.synonyms.length) sec.append(el('p', 'pos', 'مترادفات'), el('p', null, l.synonyms.join('، ')));
|
||||
if (l.ety) sec.append(el('p', 'pos', 'اشتقاق'), el('p', 'en ety', l.ety, { dir: 'ltr', lang: 'en' }));
|
||||
for (const [label, items] of [['مترادفات', l.synonyms], ['مشتقات', l.derived], ['متعلقہ الفاظ', l.related]]) {
|
||||
if (!items.length) continue;
|
||||
const box = buttons(items.map((t) => ({ title: t, langs: [l.code] })));
|
||||
sec.append(...(items.length > 8 ? [more(`${label} (${items.length})`, box)] : [el('p', 'pos', label), box]));
|
||||
}
|
||||
for (const ety of l.etymologies) {
|
||||
const text = el('p', 'en ety', ety, { dir: 'ltr', lang: 'en' });
|
||||
sec.append(...(ety.length > 220 ? [more('اشتقاق', text)] : [el('p', 'pos', 'اشتقاق'), text]));
|
||||
}
|
||||
if (l.en) {
|
||||
const p = el('p', 'pos');
|
||||
p.append(el('a', null, 'انگریزی ویکی لغت', { href: l.en, target: '_blank', rel: 'noopener' }));
|
||||
|
||||
@ -146,6 +146,10 @@ h1 + .muted { text-align: center; margin-top: 0; }
|
||||
.dict .pos { font-size: .8rem; color: var(--muted); margin: 8px 0 0; }
|
||||
.dict .note { font-size: .8rem; margin: 0; }
|
||||
.dict .via { margin: 2px 0; }
|
||||
.dict details.more { margin: 4px 0; }
|
||||
.dict details.more summary { cursor: pointer; font-size: .8rem; color: var(--brand); }
|
||||
.dict .example { font-size: .8rem; color: var(--muted); margin: 4px 0; border-left: 2px solid var(--gold-light); padding-left: 8px; }
|
||||
.dict .example .ex-text { font-family: 'Noto Naskh Arabic', serif; font-size: .95rem; color: var(--ink); }
|
||||
.dict .reading { border-inline-start: 2px solid var(--gold-light); padding-inline-start: 10px; margin: 8px 0; }
|
||||
.dict .form { font-size: 1.15rem; color: var(--lapis); }
|
||||
.dict .via .en { font-size: .78rem; }
|
||||
|
||||
Loading…
Reference in New Issue
Block a user