Word sidebar: readings by vowels, vowel variants, similar words; Persian/Arabic explained in Urdu

- readings: one spelling, different short vowels (ملک: مَلَک، مَلِک، مِلک، مُلک), each with form,
  transliteration, IPA, audio and meanings
- always: same consonants with other long vowels (consonant-skeleton regex)
- not found: inflection stems (آنکھوں -> آنکھ, جاتے -> جانا), prefix regex, trigram similarity
- Persian/Arabic words with no Urdu entry: Urdu translations via English first, then English, then
  their own language; filler glosses (especially, usually…) dropped
- punctuation, dashes and spaces dropped from selected words and lookup keys; clean titles

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Anas Rashid 2026-10-08 22:07:10 +02:00
parent 7fcc0ee7fa
commit 19b68425f6
6 changed files with 187 additions and 55 deletions

View File

@ -21,12 +21,13 @@ test('kaikki entry: glosses, IPA, audio, romanisation, form-of', () => {
forms: [{ form: 'قِسْمَت', tags: ['canonical'] }, { form: 'qismat', tags: ['romanization'] }], forms: [{ form: 'قِسْمَت', tags: ['canonical'] }, { form: 'qismat', tags: ['romanization'] }],
}); });
assert.deepEqual(e, { pos: 'noun', glosses: ['fate, destiny', 'division'], formOf: false, ipa: ['/qɪs.mət̪/'], assert.deepEqual(e, { pos: 'noun', glosses: ['fate, destiny', 'division'], formOf: false, ipa: ['/qɪs.mət̪/'],
audio: 'https://upload.wikimedia.org/x.mp3', tr: 'qismat', ety: 'From Arabic', synonyms: [] }); audio: 'https://upload.wikimedia.org/x.mp3', tr: 'qismat', form: 'قِسْمَت', ety: 'From Arabic', synonyms: [] });
assert.equal(kaikkiEntry({ word: 'x', pos: 'noun', senses: [{ glosses: ['plural of y'], tags: ['form-of'] }] }).formOf, true); assert.equal(kaikkiEntry({ word: 'x', pos: 'noun', senses: [{ glosses: ['plural of y'], tags: ['form-of'] }] }).formOf, true);
}); });
test('pivot keys from English glosses', () => { test('pivot keys from English glosses', () => {
assert.deepEqual(glossKeys(['fate, destiny', 'to love (someone)', 'a very long description of something that is not a key']), ['fate', 'destiny', 'love']); assert.deepEqual(glossKeys(['fate, destiny', 'to love (someone)', 'a very long description of something that is not a key']), ['fate', 'destiny', 'love']);
assert.deepEqual(glossKeys(['zephyr; soft breeze; especially, a morning breeze']), ['zephyr', 'soft breeze', 'morning breeze']);
}); });
test('ur.wiktionary entry: meanings and origin', () => { test('ur.wiktionary entry: meanings and origin', () => {
@ -65,3 +66,24 @@ test('etymology drops the "Etymology tree" summary', () => {
const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'Etymology tree Arabic قَسَمَ (qasama)bor. Urdu قِسْمَت Borrowed from Classical Persian قِسْمَت (qismat).' }).ety; const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'Etymology tree Arabic قَسَمَ (qasama)bor. Urdu قِسْمَت Borrowed from Classical Persian قِسْمَت (qismat).' }).ety;
assert.equal(ety, 'Borrowed from Classical Persian قِسْمَت (qismat).'); assert.equal(ety, 'Borrowed from Classical Persian قِسْمَت (qismat).');
}); });
test('similar words: punctuation-free keys, inflection stems, consonant skeleton regex', async () => {
const { stems, skeleton } = await import('./dictionary.ts');
assert.equal(key('بے ثبوت'), key('بےثبوت'));
assert.equal(key('دل،'), key('دل'));
assert.equal(key('دل-جان۔'), key('دلجان'));
assert.ok(stems(key('آنکھوں')).includes(key('آنکھ')));
assert.ok(stems(key('دیوانے')).includes(key('دیوانہ')));
assert.ok(stems(key('جاتے')).includes(key('جانا')));
const re = new RegExp(skeleton(key('دیوانگی'))!);
assert.ok(re.test(key('دیوانگی')) && re.test(key('دیونگی')) && !re.test(key('دیوار')));
assert.equal(skeleton(key('آ')), null);
});
test('suggestion titles are shown clean', async () => {
const { clean } = await import('./dictionary.ts');
assert.equal(clean('ج.ا'), 'جا');
assert.equal(clean('ـجات'), 'جات');
assert.equal(clean('دیوانه‌تر'), 'دیوانهتر');
assert.equal(clean('بے ثبوت،'), 'بےثبوت');
});

View File

@ -17,12 +17,16 @@ const KAIKKI = (l: string) => `https://kaikki.org/dictionary/${NAME[l as 'ur']}/
const DUMP = (l: string) => `https://dumps.wikimedia.org/${l}wiktionary/latest/${l}wiktionary-latest-pages-articles.xml.bz2`; const DUMP = (l: string) => `https://dumps.wikimedia.org/${l}wiktionary/latest/${l}wiktionary-latest-pages-articles.xml.bz2`;
const MARKS = /[ؐ-ًؚ-ٰٟۖ-ۭـ‌‍‎‏]/g; const MARKS = /[ؐ-ًؚ-ٰٟۖ-ۭـ‌‍‎‏]/g;
// spelling-insensitive key shared by Urdu, Persian and Arabic: no diacritics/tatweel, one heh, one yeh, one kaf, // punctuation, dashes and spaces: dropped from words (بے ثبوت = بےثبوت, دل، = دل)
// plain alef, noon ghunna as noon (ناداں = نادان, قسمت = قِسْمَت, نگاہ = نگاه, معنی = معنى). ے and ھ stay distinct. export const PUNCT = /[\s\-‐-―_.,،۔؛؟!?:;"'«»()[\]{}/\\*]+/g;
// spelling-insensitive key shared by Urdu, Persian and Arabic: no diacritics/tatweel/punctuation/spaces, one heh,
// one yeh, one kaf, plain alef, noon ghunna as noon (ناداں = نادان, قسمت = قِسْمَت, نگاہ = نگاه, معنی = معنى).
// ے and ھ stay distinct.
export const key = (w: string) => export const key = (w: string) =>
w.normalize('NFC').replace(MARKS, '') w.normalize('NFC').replace(MARKS, '').replace(PUNCT, '')
.replace(/[ہۂۃةه]/g, 'ه').replace(/[يىی]/g, 'ی') .replace(/[ہۂۃةه]/g, 'ه').replace(/[يىی]/g, 'ی')
.replace(/ك/g, 'ک').replace(/[أإٱ]/g, 'ا').replace(/ں/g, 'ن').trim(); .replace(/ك/g, 'ک').replace(/[أإٱ]/g, 'ا').replace(/ں/g, 'ن');
const unlink = (wt: string) => wt.replace(/\[\[(?:[^\]|]*\|)?([^\]]*)\]\]/g, '$1').replace(/'''?/g, '').trim(); const unlink = (wt: string) => wt.replace(/\[\[(?:[^\]|]*\|)?([^\]]*)\]\]/g, '$1').replace(/'''?/g, '').trim();
const upload = (u?: string) => (u && u.startsWith('https://upload.wikimedia.org/') ? u : null); const upload = (u?: string) => (u && u.startsWith('https://upload.wikimedia.org/') ? u : null);
@ -40,6 +44,7 @@ export function kaikkiEntry(d: any) {
ipa: [...new Set(sounds.map((s: any) => s.ipa).filter(Boolean))].slice(0, 2) as string[], ipa: [...new Set(sounds.map((s: any) => s.ipa).filter(Boolean))].slice(0, 2) as string[],
audio: sounds.map((s: any) => upload(s.mp3_url) ?? upload(s.ogg_url)).find(Boolean) ?? null, audio: sounds.map((s: any) => upload(s.mp3_url) ?? upload(s.ogg_url)).find(Boolean) ?? null,
tr: d.forms?.find((f: any) => f.tags?.includes('romanization'))?.form ?? null, tr: d.forms?.find((f: any) => f.tags?.includes('romanization'))?.form ?? null,
form: d.forms?.find((f: any) => f.tags?.includes('canonical'))?.form ?? null, // with short vowels: مُلْک
ety: d.etymology_text ? etymology(String(d.etymology_text)) : null, ety: d.etymology_text ? etymology(String(d.etymology_text)) : null,
synonyms: (d.synonyms ?? []).map((s: any) => s.word).filter(Boolean).slice(0, 8) as string[], synonyms: (d.synonyms ?? []).map((s: any) => s.word).filter(Boolean).slice(0, 8) as string[],
}; };
@ -53,7 +58,8 @@ const etymology = (t: string) =>
// English glosses of an Urdu entry as pivot keys: "fate, destiny" -> fate, destiny (short, lowercase) // English glosses of an Urdu entry as pivot keys: "fate, destiny" -> fate, destiny (short, lowercase)
export const glossKeys = (glosses: string[]) => [...new Set(glosses.flatMap((g) => g.replace(/\([^)]*\)/g, '').split(/[,;]/)) export const glossKeys = (glosses: string[]) => [...new Set(glosses.flatMap((g) => g.replace(/\([^)]*\)/g, '').split(/[,;]/))
.map((s) => s.trim().toLowerCase().replace(/^(to|a|an|the) /, '')).filter((s) => /^[a-z][a-z' -]{1,30}$/.test(s) && s.split(' ').length <= 3))]; .map((s) => s.trim().toLowerCase().replace(/^(to|a|an|the) /, '')).filter((s) => /^[a-z][a-z' -]{1,30}$/.test(s) && s.split(' ').length <= 3 && !FILLER.has(s)))];
const FILLER = new Set(['especially', 'usually', 'often', 'also', 'etc', 'something', 'someone', 'chiefly', 'mainly', 'figuratively', 'literally', 'rare', 'archaic', 'obsolete', 'dated', 'poetic', 'colloquial', 'informal', 'formal', 'slang', 'by extension']);
// ur.wiktionary Urdu entries: numbered lines under ==معانی==, else "# " lines, else the opening prose; // ur.wiktionary Urdu entries: numbered lines under ==معانی==, else "# " lines, else the opening prose;
// origin from "(عربی)" on the first line // origin from "(عربی)" on the first line
@ -280,30 +286,98 @@ export async function lookup(word: string) {
const en = rows.filter((r) => r.lang === code && r.source === 'en'); const en = rows.filter((r) => r.lang === code && r.source === 'en');
const own = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.defs.length); const own = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.defs.length);
const lemmas = en.filter((r) => !r.data.formOf), use = lemmas.length ? lemmas : en; const lemmas = en.filter((r) => !r.data.formOf), use = lemmas.length ? lemmas : en;
const first = (f: string) => use.map((r) => r.data[f]).find((v) => (Array.isArray(v) ? v.length : v)) ?? null; // readings: one spelling, different (unwritten) short vowels, e.g. ملک = مُلْک mulk, مَلِک malik, مِلْک milk
// ur.wiktionary's English translation lines ("انگریزی : …") const readings: any[] = [];
const translated = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.english).map((r) => r.data.english); for (const r of use) {
const id = r.data.tr ?? r.data.form ?? r.title;
let g = readings.find((x) => x.id === id);
if (!g) readings.push((g = { id, form: r.data.form ?? r.title, tr: r.data.tr, ipa: r.data.ipa, audio: r.data.audio, senses: [] }));
if (!g.ipa.length) g.ipa = r.data.ipa;
g.audio ??= r.data.audio;
if (r.data.glosses.length && g.senses.length < 3) g.senses.push({ pos: r.data.pos, defs: r.data.glosses.slice(0, 4) });
}
return { return {
code, ipa: first('ipa') ?? [], tr: first('tr'), audio: first('audio'), ety: first('ety'), code, readings: readings.slice(0, 5).map(({ id, ...x }) => x),
ety: use.map((r) => r.data.ety).find(Boolean) ?? null,
synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))].slice(0, 8), synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))].slice(0, 8),
meanings: own.flatMap((r) => r.data.defs).slice(0, 6), origin: own.find((r) => r.data.origin)?.data.origin ?? null, meanings: own.flatMap((r) => r.data.defs).slice(0, 6), origin: own.find((r) => r.data.origin)?.data.origin ?? null,
source: own[0] ? page(code, 'own', own[0].title) : null, source: own[0] ? page(code, 'own', own[0].title) : null,
senses: [...use.map((r) => ({ pos: r.data.pos, defs: r.data.glosses.slice(0, 4) })).filter((s) => s.defs.length).slice(0, 3), // ur.wiktionary's English translation lines ("انگریزی : …")
...(translated.length ? [{ pos: 'translation', defs: translated }] : [])], translation: rows.filter((r) => r.lang === code && r.source === 'own' && r.data.english).map((r) => r.data.english),
en: en[0] ? page(code, 'en', en[0].title) : null, en: en[0] ? page(code, 'en', en[0].title) : null,
}; };
}).filter((l) => l.meanings.length || l.senses.length || l.ipa.length || l.tr); }).filter((l) => l.meanings.length || l.readings.length || l.translation.length);
// no Urdu meaning: Urdu words sharing the primary English meaning // Urdu translations through English (the primary English meaning's Urdu words): for the Urdu entry when it
let equivalents: { gloss: string; urdu: string[] }[] = []; // has no Urdu-language meaning, and for the Persian and Arabic entries when the word is not in Urdu at all
if (!langs.some((l) => l.code === 'ur' && l.meanings.length)) { const inUrdu = langs.some((l) => l.code === 'ur');
const glosses = glossKeys(langs.find((l) => l.senses.length)?.senses[0].defs.slice(0, 2) ?? []).slice(0, 4); await Promise.all(langs.map(async (l: any) => {
if (glosses.length) { l.urdu = (l.code === 'ur' ? !l.meanings.length : !inUrdu) ? await viaEnglish(l.readings[0]?.senses[0]?.defs.slice(0, 2) ?? [], k) : [];
const { rows: eq } = await pool.query('SELECT gloss, word FROM ur_glosses WHERE gloss = ANY($1)', [glosses]); }));
const seen = new Set([k]); const found = langs.length > 0;
equivalents = glosses.map((g) => ({ // always: the same consonants with other long vowels (دل: دال، دول، دیل); not found: also stems and near spellings
gloss: g, urdu: eq.filter((r) => r.gloss === g && !seen.has(key(r.word)) && seen.add(key(r.word))).map((r) => r.word).slice(0, 8), const [variants, near] = await Promise.all([vowelVariants(k), found ? [] : similar(k)]);
})).filter((e) => e.urdu.length); return { word, langs, found, variants, similar: near.filter((n) => !variants.some((v) => v.title === n.title)) };
} }
}
return { word, langs, equivalents, found: langs.length > 0 }; async function viaEnglish(glosses: string[], k: string) {
const g = glossKeys(glosses).slice(0, 4);
if (!g.length) return [];
const { rows } = await pool.query('SELECT gloss, word FROM ur_glosses WHERE gloss = ANY($1)', [g]);
const seen = new Set([k]);
return g.map((gloss) => ({
gloss, urdu: rows.filter((r) => r.gloss === gloss && !seen.has(key(r.word)) && seen.add(key(r.word))).map((r) => r.word).slice(0, 8),
})).filter((e) => e.urdu.length);
}
// ---- not found: similar words ----
// Urdu inflection endings (in key form) and the base forms to try: آنکھوں -> آنکھ, دیوانے -> دیوانه, جاتے -> جانا
const ENDINGS = ['یاں', 'ئیں', 'ؤں', 'وں', 'یں', 'گی', 'گا', 'گے', 'تا', 'تے', 'تی', 'نا', 'نے', 'نی', 'ے', 'ی', 'ا', 'و', 'ه'].map(key); // in key form (ں -> ن)
export function stems(k: string) {
const out = new Set<string>();
for (const e of ENDINGS) {
if (!k.endsWith(e) || k.length - e.length < 2) continue;
const base = k.slice(0, -e.length);
for (const s of [base, base + 'ه', base + 'ا', base + 'نا']) if (s !== k) out.add(s);
}
return [...out];
}
// consonant skeleton as a regex: long vowels, heh and hamza optional (spellings add or drop them)
const SOFT = 'اوی\u0647ےءئ';
const esc = (c: string) => c.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
export function skeleton(k: string) {
const cs = [...k].filter((c) => !SOFT.includes(c));
if (cs.length < 2) return null;
const v = `[${SOFT}]*`;
return `^${v}${cs.map(esc).join(v)}${v}$`;
}
// shown titles without punctuation, dashes, spaces, tatweel or zero-width joiners
export const clean = (t: string) => t.replace(PUNCT, '').replace(/[\u0640\u200C\u200D]/g, '');
async function vowelVariants(k: string) {
const sk = skeleton(k);
if (!sk) return [];
const { rows } = await pool.query(
`SELECT key, (array_agg(title ORDER BY lang <> 'ur', lang <> 'fa', title))[1] AS title, array_agg(DISTINCT lang) AS langs
FROM wiktionary WHERE key ~ $1 AND key <> $2 GROUP BY key ORDER BY bool_or(lang = 'ur') DESC, similarity(key, $2) DESC LIMIT 10`, [sk, k]);
return rows.map((r) => ({ title: clean(r.title), langs: LANGS.filter((l) => r.langs.includes(l)) }));
}
async function similar(k: string) {
const st = stems(k), sk = skeleton(k), pre = [k, ...st].filter((s) => s.length >= 3).sort((a, b) => b.length - a.length)[0];
// one row per key: the Urdu spelling when there is one; Urdu entries first within each tier
const pick = `SELECT key, (array_agg(title ORDER BY lang <> 'ur', lang <> 'fa', title))[1] AS title, array_agg(DISTINCT lang) AS langs FROM wiktionary`;
const ur = `bool_or(lang = 'ur') DESC`;
const [a, b, c, d] = await Promise.all([
st.length ? pool.query(`${pick} WHERE key = ANY($1) GROUP BY key ORDER BY ${ur}`, [st]) : { rows: [] },
sk ? pool.query(`${pick} WHERE key ~ $1 AND key <> $2 GROUP BY key ORDER BY ${ur}, similarity(key, $2) DESC LIMIT 8`, [sk, k]) : { rows: [] },
pre ? pool.query(`${pick} WHERE key ~ $1 AND key <> $2 GROUP BY key ORDER BY ${ur}, length(key) LIMIT 8`, ['^' + [...pre].map(esc).join(''), k]) : { rows: [] },
pool.query(`${pick} WHERE key % $1 AND key <> $1 GROUP BY key ORDER BY ${ur}, similarity(key, $1) DESC LIMIT 8`, [k]),
]);
const seen = new Set<string>(), out: { title: string; langs: string[] }[] = [];
for (const r of [...a.rows, ...b.rows, ...c.rows, ...d.rows])
if (!seen.has(r.key) && seen.add(r.key)) out.push({ title: clean(r.title), langs: LANGS.filter((l) => r.langs.includes(l)) });
return out.slice(0, 12);
} }

View File

@ -7,7 +7,7 @@
import Fastify from 'fastify'; import Fastify from 'fastify';
import { pool } from './db.ts'; import { pool } from './db.ts';
import { likePatterns, normalise, terms } from './urdu.ts'; import { likePatterns, normalise, terms } from './urdu.ts';
import { lookup } from './dictionary.ts'; import { lookup, PUNCT } from './dictionary.ts';
const app = Fastify({ logger: { level: process.env.LOG_LEVEL ?? 'info' } }); const app = Fastify({ logger: { level: process.env.LOG_LEVEL ?? 'info' } });
const PAGE_SIZE = 20; const PAGE_SIZE = 20;
@ -128,7 +128,7 @@ app.get<{ Querystring: { q?: string; poet?: string; page?: string } }>('/api/sea
// one word in Arabic script (Urdu, Persian, Arabic), as selected by a reader // one word in Arabic script (Urdu, Persian, Arabic), as selected by a reader
app.get<{ Querystring: { w?: string } }>('/api/word', async (req, reply) => { app.get<{ Querystring: { w?: string } }>('/api/word', async (req, reply) => {
const w = (req.query.w ?? '').trim(); const w = (req.query.w ?? '').replace(PUNCT, ''); // commas, dots, dashes, spaces
if (!/^[\p{Script=Arabic}\p{M}\u200C]{1,40}$/u.test(w)) return reply.code(400).send({ error: 'one Urdu, Persian or Arabic word' }); if (!/^[\p{Script=Arabic}\p{M}\u200C]{1,40}$/u.test(w)) return reply.code(400).send({ error: 'one Urdu, Persian or Arabic word' });
return lookup(w); return lookup(w);
}); });

View File

@ -70,6 +70,7 @@ CREATE TABLE IF NOT EXISTS wiktionary (
); );
CREATE INDEX IF NOT EXISTS wiktionary_key ON wiktionary(key); CREATE INDEX IF NOT EXISTS wiktionary_key ON wiktionary(key);
CREATE INDEX IF NOT EXISTS wiktionary_title ON wiktionary(lang, source, title); CREATE INDEX IF NOT EXISTS wiktionary_title ON wiktionary(lang, source, title);
CREATE INDEX IF NOT EXISTS wiktionary_key_trgm ON wiktionary USING gin (key gin_trgm_ops); -- similar words (regex, %)
CREATE TABLE IF NOT EXISTS ur_glosses ( -- English gloss -> Urdu word, for the pivot through English CREATE TABLE IF NOT EXISTS ur_glosses ( -- English gloss -> Urdu word, for the pivot through English
gloss text NOT NULL, gloss text NOT NULL,
word text NOT NULL word text NOT NULL

View File

@ -136,44 +136,72 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
dictBody.replaceChildren(el('h2', null, w), el('p', 'muted', 'تلاش جاری ہے…')); dictBody.replaceChildren(el('h2', null, w), el('p', 'muted', 'تلاش جاری ہے…'));
let d = null; let d = null;
// v: bump when the response format changes (responses are browser-cached for a day) // v: bump when the response format changes (responses are browser-cached for a day)
try { const r = await fetch('/api/word?v=2&w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {} try { const r = await fetch('/api/word?v=5&w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {}
if (w !== current) return; // a newer selection won if (w !== current) return; // a newer selection won
const words = (title, items, into) => { // clickable word buttons
if (!items.length) return;
const box = el('div', 'similar');
for (const s of items) {
const b = el('button', null, s.title, { type: 'button', title: s.langs.map((c) => LANG[c]).join('، ') });
b.onclick = () => showWord(s.title);
box.append(b);
}
into.push(el('h3', null, title), box);
};
const list = (items, cls, attrs) => { const ol = el('ol', cls, null, attrs); items.forEach((x) => ol.append(el('li', null, x))); return ol; };
const out = [el('h2', null, w)]; const out = [el('h2', null, w)];
if (!d || !d.found) { if (!d || !d.found) {
dictBody.replaceChildren(...out, el('p', 'muted', 'ویکی لغت میں یہ لفظ نہیں ملا۔')); out.push(el('p', 'muted', 'ویکی لغت میں یہ لفظ نہیں ملا۔'));
words('ملتے جلتے الفاظ', [...(d?.similar ?? []), ...(d?.variants ?? [])], out);
dictBody.replaceChildren(...out);
return; return;
} }
const list = (items, cls, attrs) => { const ol = el('ol', cls, null, attrs); items.forEach((x) => ol.append(el('li', null, x))); return ol; }; // one section per language, in order اردو، فارسی، عربی; each says where its meanings come from.
const equivalents = () => { // A Persian or Arabic word is explained first by its Urdu translations, then English, then its own language.
const sec = el('section', 'dict-lang');
sec.append(el('h3', null, 'ہم معنی اردو الفاظ'), el('p', 'muted note', 'انگریزی کے واسطے سے'));
for (const e of d.equivalents) {
const p = el('p');
p.append(el('bdi', 'en', e.gloss), ': ' + e.urdu.join('، '));
sec.append(p);
}
return sec;
};
// one section per language, in order اردو، فارسی، عربی; each says where its meanings come from
for (const l of d.langs) { for (const l of d.langs) {
const sec = el('section', 'dict-lang'); const sec = el('section', 'dict-lang');
sec.append(el('h3', null, LANG[l.code])); sec.append(el('h3', null, LANG[l.code]));
// a pronunciation line: vowelled form, transliteration, IPA, audio
const pron = (r) => {
const pr = el('p', 'pron'); const pr = el('p', 'pron');
if (l.tr) pr.append(el('bdi', 'en', l.tr)); if (r.form) pr.append(el('span', 'form', r.form), ' ');
l.ipa.forEach((i) => pr.append(' ', el('bdi', 'ipa', i))); if (r.tr) pr.append(el('bdi', 'en', r.tr));
if (l.audio && l.audio.startsWith('https://upload.wikimedia.org/')) { r.ipa.forEach((i) => pr.append(' ', el('bdi', 'ipa', i)));
if (r.audio && r.audio.startsWith('https://upload.wikimedia.org/')) {
const b = el('button', 'play', '🔊', { type: 'button', 'aria-label': 'تلفظ سنیں' }); const b = el('button', 'play', '🔊', { type: 'button', 'aria-label': 'تلفظ سنیں' });
b.onclick = () => new Audio(l.audio).play(); b.onclick = () => new Audio(r.audio).play();
pr.append(' ', b); pr.append(' ', b);
} }
if (pr.childNodes.length) sec.append(pr); return pr;
};
const urdu = () => {
if (!l.urdu?.length) return;
sec.append(el('p', 'pos', 'اردو میں · انگریزی ویکی لغت کے واسطے سے'));
for (const e of l.urdu) {
const p = el('p', 'via');
p.append(e.urdu.join('، '), ' ', el('bdi', 'en muted', `(${e.gloss})`));
sec.append(p);
}
};
// readings: the same spelling with different short vowels (ملک: مُلک، مَلِک، مِلک), each with its meanings
const english = () => {
if (l.readings.length > 1) sec.append(el('p', 'pos', `${l.readings.length} قراءتیں (حرکات کے فرق سے)`));
for (const r of l.readings) {
const box = el('div', l.readings.length > 1 ? 'reading' : null);
box.append(pron(r));
for (const s of r.senses) box.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos.charAt(0).toUpperCase() + s.pos.slice(1)] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' }));
sec.append(box);
}
if (l.translation.length) sec.append(el('p', 'pos', POS.Translation), list(l.translation, 'en', { dir: 'ltr', lang: 'en' }));
};
if (l.code !== 'ur') { urdu(); english(); }
if (l.meanings.length) { if (l.meanings.length) {
const label = el('p', 'pos'); const label = el('p', 'pos');
label.append(`معانی: ${LANG[l.code]} ویکی لغت` + (l.origin ? ` · اصل: ${l.origin}` : '') + ' · '); label.append(`معانی: ${LANG[l.code]} ویکی لغت` + (l.origin ? ` · اصل: ${l.origin}` : '') + ' · ');
if (l.source) label.append(el('a', null, 'ماخذ', { href: l.source, target: '_blank', rel: 'noopener' })); if (l.source) label.append(el('a', null, 'ماخذ', { href: l.source, target: '_blank', rel: 'noopener' }));
sec.append(label, list(l.meanings, null, { lang: l.code })); sec.append(label, list(l.meanings, null, { lang: l.code }));
} }
for (const s of l.senses) sec.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos.charAt(0).toUpperCase() + s.pos.slice(1)] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' })); if (l.code === 'ur') { english(); urdu(); }
if (l.synonyms.length) sec.append(el('p', 'pos', 'مترادفات'), el('p', null, l.synonyms.join('، '))); if (l.synonyms.length) sec.append(el('p', 'pos', 'مترادفات'), el('p', null, l.synonyms.join('، ')));
if (l.ety) sec.append(el('p', 'pos', 'اشتقاق'), el('p', 'en ety', l.ety, { dir: 'ltr', lang: 'en' })); if (l.ety) sec.append(el('p', 'pos', 'اشتقاق'), el('p', 'en ety', l.ety, { dir: 'ltr', lang: 'en' }));
if (l.en) { if (l.en) {
@ -182,15 +210,14 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
sec.append(p); sec.append(p);
} }
out.push(sec); out.push(sec);
if (l.code === 'ur' && d.equivalents.length) out.push(equivalents());
} }
if (d.equivalents.length && !d.langs.some((l) => l.code === 'ur')) out.splice(1, 0, equivalents()); words('ملتے جلتے الفاظ · دیگر حروفِ علت', d.variants ?? [], out); // same consonants, other long vowels
const src = el('p', 'src muted'); const src = el('p', 'src muted');
src.append('ویکی لغت (Wiktionary) · '); src.append('ویکی لغت (Wiktionary) · ');
src.append('CC BY-SA'); src.append('CC BY-SA');
dictBody.replaceChildren(...out, src); dictBody.replaceChildren(...out, src);
} }
const TRIM = /^[\s"'«»()\[\]،۔؟؛!:.,]+|[\s"'«»()\[\]،۔؟؛!:.,]+$/g; const TRIM = /["'«»()\[\]،۔؟؛!:.,\-–—_]+|^\s+|\s+$/g; // punctuation and dashes anywhere, outer spaces
let selTimer; let selTimer;
document.addEventListener('selectionchange', () => { document.addEventListener('selectionchange', () => {
clearTimeout(selTimer); clearTimeout(selTimer);
@ -198,6 +225,7 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
const sel = getSelection(); const sel = getSelection();
if (!sel || sel.isCollapsed || !main.contains(sel.anchorNode)) return; if (!sel || sel.isCollapsed || !main.contains(sel.anchorNode)) return;
const w = sel.toString().replace(TRIM, '').replace(/\u0614/g, ''); // drop the takhallus sign const w = sel.toString().replace(TRIM, '').replace(/\u0614/g, ''); // drop the takhallus sign
if (/\s/.test(w)) return; // one word only
if (/^[\p{Script=Arabic}\p{M}\u200C]{1,40}$/u.test(w)) showWord(w); if (/^[\p{Script=Arabic}\p{M}\u200C]{1,40}$/u.test(w)) showWord(w);
}, 350); }, 350);
}); });

View File

@ -145,9 +145,16 @@ h1 + .muted { text-align: center; margin-top: 0; }
.dict .pron { margin: 2px 0; } .dict .pron { margin: 2px 0; }
.dict .pos { font-size: .8rem; color: var(--muted); margin: 8px 0 0; } .dict .pos { font-size: .8rem; color: var(--muted); margin: 8px 0 0; }
.dict .note { font-size: .8rem; margin: 0; } .dict .note { font-size: .8rem; margin: 0; }
.dict .via { margin: 2px 0; }
.dict .reading { border-inline-start: 2px solid var(--gold-light); padding-inline-start: 10px; margin: 8px 0; }
.dict .form { font-size: 1.15rem; color: var(--lapis); }
.dict .via .en { font-size: .78rem; }
.dict .src { font-size: .78rem; margin-top: 18px; } .dict .src { font-size: .78rem; margin-top: 18px; }
.dict .ety { text-align: left; color: var(--muted); font-size: .8rem; } .dict .ety { text-align: left; color: var(--muted); font-size: .8rem; }
.dict-close { position: sticky; top: 0; float: left; border: 0; background: none; color: var(--muted); font-size: 1rem; cursor: pointer; } .dict-close { position: sticky; top: 0; float: left; border: 0; background: none; color: var(--muted); font-size: 1rem; cursor: pointer; }
.dict .similar { display: flex; flex-wrap: wrap; gap: 6px; }
.dict .similar button { font: inherit; border: 1px solid var(--border); background: var(--inner); color: var(--ink); border-radius: 999px; padding: 0 12px; cursor: pointer; }
.dict .similar button:hover { border-color: var(--brand); color: var(--brand); }
.dict .play { border: 1px solid var(--border); background: var(--inner); border-radius: 8px; cursor: pointer; padding: 0 6px; } .dict .play { border: 1px solid var(--border); background: var(--inner); border-radius: 8px; cursor: pointer; padding: 0 6px; }
/* wide screens: the page moves over instead of being covered */ /* wide screens: the page moves over instead of being covered */
@media (min-width: 1100px) { body.dict-open { padding-left: min(340px, 92vw); } } @media (min-width: 1100px) { body.dict-open { padding-left: min(340px, 92vw); } }