Word sidebar: readings by vowels, vowel variants, similar words; Persian/Arabic explained in Urdu
- readings: one spelling, different short vowels (ملک: مَلَک، مَلِک، مِلک، مُلک), each with form, transliteration, IPA, audio and meanings - always: same consonants with other long vowels (consonant-skeleton regex) - not found: inflection stems (آنکھوں -> آنکھ, جاتے -> جانا), prefix regex, trigram similarity - Persian/Arabic words with no Urdu entry: Urdu translations via English first, then English, then their own language; filler glosses (especially, usually…) dropped - punctuation, dashes and spaces dropped from selected words and lookup keys; clean titles Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
7fcc0ee7fa
commit
19b68425f6
@ -21,12 +21,13 @@ test('kaikki entry: glosses, IPA, audio, romanisation, form-of', () => {
|
|||||||
forms: [{ form: 'قِسْمَت', tags: ['canonical'] }, { form: 'qismat', tags: ['romanization'] }],
|
forms: [{ form: 'قِسْمَت', tags: ['canonical'] }, { form: 'qismat', tags: ['romanization'] }],
|
||||||
});
|
});
|
||||||
assert.deepEqual(e, { pos: 'noun', glosses: ['fate, destiny', 'division'], formOf: false, ipa: ['/qɪs.mət̪/'],
|
assert.deepEqual(e, { pos: 'noun', glosses: ['fate, destiny', 'division'], formOf: false, ipa: ['/qɪs.mət̪/'],
|
||||||
audio: 'https://upload.wikimedia.org/x.mp3', tr: 'qismat', ety: 'From Arabic', synonyms: [] });
|
audio: 'https://upload.wikimedia.org/x.mp3', tr: 'qismat', form: 'قِسْمَت', ety: 'From Arabic', synonyms: [] });
|
||||||
assert.equal(kaikkiEntry({ word: 'x', pos: 'noun', senses: [{ glosses: ['plural of y'], tags: ['form-of'] }] }).formOf, true);
|
assert.equal(kaikkiEntry({ word: 'x', pos: 'noun', senses: [{ glosses: ['plural of y'], tags: ['form-of'] }] }).formOf, true);
|
||||||
});
|
});
|
||||||
|
|
||||||
test('pivot keys from English glosses', () => {
|
test('pivot keys from English glosses', () => {
|
||||||
assert.deepEqual(glossKeys(['fate, destiny', 'to love (someone)', 'a very long description of something that is not a key']), ['fate', 'destiny', 'love']);
|
assert.deepEqual(glossKeys(['fate, destiny', 'to love (someone)', 'a very long description of something that is not a key']), ['fate', 'destiny', 'love']);
|
||||||
|
assert.deepEqual(glossKeys(['zephyr; soft breeze; especially, a morning breeze']), ['zephyr', 'soft breeze', 'morning breeze']);
|
||||||
});
|
});
|
||||||
|
|
||||||
test('ur.wiktionary entry: meanings and origin', () => {
|
test('ur.wiktionary entry: meanings and origin', () => {
|
||||||
@ -65,3 +66,24 @@ test('etymology drops the "Etymology tree" summary', () => {
|
|||||||
const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'Etymology tree Arabic قَسَمَ (qasama)bor. Urdu قِسْمَت Borrowed from Classical Persian قِسْمَت (qismat).' }).ety;
|
const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'Etymology tree Arabic قَسَمَ (qasama)bor. Urdu قِسْمَت Borrowed from Classical Persian قِسْمَت (qismat).' }).ety;
|
||||||
assert.equal(ety, 'Borrowed from Classical Persian قِسْمَت (qismat).');
|
assert.equal(ety, 'Borrowed from Classical Persian قِسْمَت (qismat).');
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test('similar words: punctuation-free keys, inflection stems, consonant skeleton regex', async () => {
|
||||||
|
const { stems, skeleton } = await import('./dictionary.ts');
|
||||||
|
assert.equal(key('بے ثبوت'), key('بےثبوت'));
|
||||||
|
assert.equal(key('دل،'), key('دل'));
|
||||||
|
assert.equal(key('دل-جان۔'), key('دلجان'));
|
||||||
|
assert.ok(stems(key('آنکھوں')).includes(key('آنکھ')));
|
||||||
|
assert.ok(stems(key('دیوانے')).includes(key('دیوانہ')));
|
||||||
|
assert.ok(stems(key('جاتے')).includes(key('جانا')));
|
||||||
|
const re = new RegExp(skeleton(key('دیوانگی'))!);
|
||||||
|
assert.ok(re.test(key('دیوانگی')) && re.test(key('دیونگی')) && !re.test(key('دیوار')));
|
||||||
|
assert.equal(skeleton(key('آ')), null);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('suggestion titles are shown clean', async () => {
|
||||||
|
const { clean } = await import('./dictionary.ts');
|
||||||
|
assert.equal(clean('ج.ا'), 'جا');
|
||||||
|
assert.equal(clean('ـجات'), 'جات');
|
||||||
|
assert.equal(clean('دیوانهتر'), 'دیوانهتر');
|
||||||
|
assert.equal(clean('بے ثبوت،'), 'بےثبوت');
|
||||||
|
});
|
||||||
|
|||||||
@ -17,12 +17,16 @@ const KAIKKI = (l: string) => `https://kaikki.org/dictionary/${NAME[l as 'ur']}/
|
|||||||
const DUMP = (l: string) => `https://dumps.wikimedia.org/${l}wiktionary/latest/${l}wiktionary-latest-pages-articles.xml.bz2`;
|
const DUMP = (l: string) => `https://dumps.wikimedia.org/${l}wiktionary/latest/${l}wiktionary-latest-pages-articles.xml.bz2`;
|
||||||
const MARKS = /[ؐ-ًؚ-ٰٟۖ-ۭـ]/g;
|
const MARKS = /[ؐ-ًؚ-ٰٟۖ-ۭـ]/g;
|
||||||
|
|
||||||
// spelling-insensitive key shared by Urdu, Persian and Arabic: no diacritics/tatweel, one heh, one yeh, one kaf,
|
// punctuation, dashes and spaces: dropped from words (بے ثبوت = بےثبوت, دل، = دل)
|
||||||
// plain alef, noon ghunna as noon (ناداں = نادان, قسمت = قِسْمَت, نگاہ = نگاه, معنی = معنى). ے and ھ stay distinct.
|
export const PUNCT = /[\s\-‐-―_.,،۔؛؟!?:;"'«»()[\]{}/\\*]+/g;
|
||||||
|
|
||||||
|
// spelling-insensitive key shared by Urdu, Persian and Arabic: no diacritics/tatweel/punctuation/spaces, one heh,
|
||||||
|
// one yeh, one kaf, plain alef, noon ghunna as noon (ناداں = نادان, قسمت = قِسْمَت, نگاہ = نگاه, معنی = معنى).
|
||||||
|
// ے and ھ stay distinct.
|
||||||
export const key = (w: string) =>
|
export const key = (w: string) =>
|
||||||
w.normalize('NFC').replace(MARKS, '')
|
w.normalize('NFC').replace(MARKS, '').replace(PUNCT, '')
|
||||||
.replace(/[ہۂۃةه]/g, 'ه').replace(/[يىی]/g, 'ی')
|
.replace(/[ہۂۃةه]/g, 'ه').replace(/[يىی]/g, 'ی')
|
||||||
.replace(/ك/g, 'ک').replace(/[أإٱ]/g, 'ا').replace(/ں/g, 'ن').trim();
|
.replace(/ك/g, 'ک').replace(/[أإٱ]/g, 'ا').replace(/ں/g, 'ن');
|
||||||
|
|
||||||
const unlink = (wt: string) => wt.replace(/\[\[(?:[^\]|]*\|)?([^\]]*)\]\]/g, '$1').replace(/'''?/g, '').trim();
|
const unlink = (wt: string) => wt.replace(/\[\[(?:[^\]|]*\|)?([^\]]*)\]\]/g, '$1').replace(/'''?/g, '').trim();
|
||||||
const upload = (u?: string) => (u && u.startsWith('https://upload.wikimedia.org/') ? u : null);
|
const upload = (u?: string) => (u && u.startsWith('https://upload.wikimedia.org/') ? u : null);
|
||||||
@ -40,6 +44,7 @@ export function kaikkiEntry(d: any) {
|
|||||||
ipa: [...new Set(sounds.map((s: any) => s.ipa).filter(Boolean))].slice(0, 2) as string[],
|
ipa: [...new Set(sounds.map((s: any) => s.ipa).filter(Boolean))].slice(0, 2) as string[],
|
||||||
audio: sounds.map((s: any) => upload(s.mp3_url) ?? upload(s.ogg_url)).find(Boolean) ?? null,
|
audio: sounds.map((s: any) => upload(s.mp3_url) ?? upload(s.ogg_url)).find(Boolean) ?? null,
|
||||||
tr: d.forms?.find((f: any) => f.tags?.includes('romanization'))?.form ?? null,
|
tr: d.forms?.find((f: any) => f.tags?.includes('romanization'))?.form ?? null,
|
||||||
|
form: d.forms?.find((f: any) => f.tags?.includes('canonical'))?.form ?? null, // with short vowels: مُلْک
|
||||||
ety: d.etymology_text ? etymology(String(d.etymology_text)) : null,
|
ety: d.etymology_text ? etymology(String(d.etymology_text)) : null,
|
||||||
synonyms: (d.synonyms ?? []).map((s: any) => s.word).filter(Boolean).slice(0, 8) as string[],
|
synonyms: (d.synonyms ?? []).map((s: any) => s.word).filter(Boolean).slice(0, 8) as string[],
|
||||||
};
|
};
|
||||||
@ -53,7 +58,8 @@ const etymology = (t: string) =>
|
|||||||
|
|
||||||
// English glosses of an Urdu entry as pivot keys: "fate, destiny" -> fate, destiny (short, lowercase)
|
// English glosses of an Urdu entry as pivot keys: "fate, destiny" -> fate, destiny (short, lowercase)
|
||||||
export const glossKeys = (glosses: string[]) => [...new Set(glosses.flatMap((g) => g.replace(/\([^)]*\)/g, '').split(/[,;]/))
|
export const glossKeys = (glosses: string[]) => [...new Set(glosses.flatMap((g) => g.replace(/\([^)]*\)/g, '').split(/[,;]/))
|
||||||
.map((s) => s.trim().toLowerCase().replace(/^(to|a|an|the) /, '')).filter((s) => /^[a-z][a-z' -]{1,30}$/.test(s) && s.split(' ').length <= 3))];
|
.map((s) => s.trim().toLowerCase().replace(/^(to|a|an|the) /, '')).filter((s) => /^[a-z][a-z' -]{1,30}$/.test(s) && s.split(' ').length <= 3 && !FILLER.has(s)))];
|
||||||
|
const FILLER = new Set(['especially', 'usually', 'often', 'also', 'etc', 'something', 'someone', 'chiefly', 'mainly', 'figuratively', 'literally', 'rare', 'archaic', 'obsolete', 'dated', 'poetic', 'colloquial', 'informal', 'formal', 'slang', 'by extension']);
|
||||||
|
|
||||||
// ur.wiktionary Urdu entries: numbered lines under ==معانی==, else "# " lines, else the opening prose;
|
// ur.wiktionary Urdu entries: numbered lines under ==معانی==, else "# " lines, else the opening prose;
|
||||||
// origin from "(عربی)" on the first line
|
// origin from "(عربی)" on the first line
|
||||||
@ -280,30 +286,98 @@ export async function lookup(word: string) {
|
|||||||
const en = rows.filter((r) => r.lang === code && r.source === 'en');
|
const en = rows.filter((r) => r.lang === code && r.source === 'en');
|
||||||
const own = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.defs.length);
|
const own = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.defs.length);
|
||||||
const lemmas = en.filter((r) => !r.data.formOf), use = lemmas.length ? lemmas : en;
|
const lemmas = en.filter((r) => !r.data.formOf), use = lemmas.length ? lemmas : en;
|
||||||
const first = (f: string) => use.map((r) => r.data[f]).find((v) => (Array.isArray(v) ? v.length : v)) ?? null;
|
// readings: one spelling, different (unwritten) short vowels, e.g. ملک = مُلْک mulk, مَلِک malik, مِلْک milk
|
||||||
// ur.wiktionary's English translation lines ("انگریزی : …")
|
const readings: any[] = [];
|
||||||
const translated = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.english).map((r) => r.data.english);
|
for (const r of use) {
|
||||||
|
const id = r.data.tr ?? r.data.form ?? r.title;
|
||||||
|
let g = readings.find((x) => x.id === id);
|
||||||
|
if (!g) readings.push((g = { id, form: r.data.form ?? r.title, tr: r.data.tr, ipa: r.data.ipa, audio: r.data.audio, senses: [] }));
|
||||||
|
if (!g.ipa.length) g.ipa = r.data.ipa;
|
||||||
|
g.audio ??= r.data.audio;
|
||||||
|
if (r.data.glosses.length && g.senses.length < 3) g.senses.push({ pos: r.data.pos, defs: r.data.glosses.slice(0, 4) });
|
||||||
|
}
|
||||||
return {
|
return {
|
||||||
code, ipa: first('ipa') ?? [], tr: first('tr'), audio: first('audio'), ety: first('ety'),
|
code, readings: readings.slice(0, 5).map(({ id, ...x }) => x),
|
||||||
|
ety: use.map((r) => r.data.ety).find(Boolean) ?? null,
|
||||||
synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))].slice(0, 8),
|
synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))].slice(0, 8),
|
||||||
meanings: own.flatMap((r) => r.data.defs).slice(0, 6), origin: own.find((r) => r.data.origin)?.data.origin ?? null,
|
meanings: own.flatMap((r) => r.data.defs).slice(0, 6), origin: own.find((r) => r.data.origin)?.data.origin ?? null,
|
||||||
source: own[0] ? page(code, 'own', own[0].title) : null,
|
source: own[0] ? page(code, 'own', own[0].title) : null,
|
||||||
senses: [...use.map((r) => ({ pos: r.data.pos, defs: r.data.glosses.slice(0, 4) })).filter((s) => s.defs.length).slice(0, 3),
|
// ur.wiktionary's English translation lines ("انگریزی : …")
|
||||||
...(translated.length ? [{ pos: 'translation', defs: translated }] : [])],
|
translation: rows.filter((r) => r.lang === code && r.source === 'own' && r.data.english).map((r) => r.data.english),
|
||||||
en: en[0] ? page(code, 'en', en[0].title) : null,
|
en: en[0] ? page(code, 'en', en[0].title) : null,
|
||||||
};
|
};
|
||||||
}).filter((l) => l.meanings.length || l.senses.length || l.ipa.length || l.tr);
|
}).filter((l) => l.meanings.length || l.readings.length || l.translation.length);
|
||||||
// no Urdu meaning: Urdu words sharing the primary English meaning
|
// Urdu translations through English (the primary English meaning's Urdu words): for the Urdu entry when it
|
||||||
let equivalents: { gloss: string; urdu: string[] }[] = [];
|
// has no Urdu-language meaning, and for the Persian and Arabic entries when the word is not in Urdu at all
|
||||||
if (!langs.some((l) => l.code === 'ur' && l.meanings.length)) {
|
const inUrdu = langs.some((l) => l.code === 'ur');
|
||||||
const glosses = glossKeys(langs.find((l) => l.senses.length)?.senses[0].defs.slice(0, 2) ?? []).slice(0, 4);
|
await Promise.all(langs.map(async (l: any) => {
|
||||||
if (glosses.length) {
|
l.urdu = (l.code === 'ur' ? !l.meanings.length : !inUrdu) ? await viaEnglish(l.readings[0]?.senses[0]?.defs.slice(0, 2) ?? [], k) : [];
|
||||||
const { rows: eq } = await pool.query('SELECT gloss, word FROM ur_glosses WHERE gloss = ANY($1)', [glosses]);
|
}));
|
||||||
const seen = new Set([k]);
|
const found = langs.length > 0;
|
||||||
equivalents = glosses.map((g) => ({
|
// always: the same consonants with other long vowels (دل: دال، دول، دیل); not found: also stems and near spellings
|
||||||
gloss: g, urdu: eq.filter((r) => r.gloss === g && !seen.has(key(r.word)) && seen.add(key(r.word))).map((r) => r.word).slice(0, 8),
|
const [variants, near] = await Promise.all([vowelVariants(k), found ? [] : similar(k)]);
|
||||||
})).filter((e) => e.urdu.length);
|
return { word, langs, found, variants, similar: near.filter((n) => !variants.some((v) => v.title === n.title)) };
|
||||||
}
|
}
|
||||||
}
|
|
||||||
return { word, langs, equivalents, found: langs.length > 0 };
|
async function viaEnglish(glosses: string[], k: string) {
|
||||||
|
const g = glossKeys(glosses).slice(0, 4);
|
||||||
|
if (!g.length) return [];
|
||||||
|
const { rows } = await pool.query('SELECT gloss, word FROM ur_glosses WHERE gloss = ANY($1)', [g]);
|
||||||
|
const seen = new Set([k]);
|
||||||
|
return g.map((gloss) => ({
|
||||||
|
gloss, urdu: rows.filter((r) => r.gloss === gloss && !seen.has(key(r.word)) && seen.add(key(r.word))).map((r) => r.word).slice(0, 8),
|
||||||
|
})).filter((e) => e.urdu.length);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- not found: similar words ----
|
||||||
|
|
||||||
|
// Urdu inflection endings (in key form) and the base forms to try: آنکھوں -> آنکھ, دیوانے -> دیوانه, جاتے -> جانا
|
||||||
|
const ENDINGS = ['یاں', 'ئیں', 'ؤں', 'وں', 'یں', 'گی', 'گا', 'گے', 'تا', 'تے', 'تی', 'نا', 'نے', 'نی', 'ے', 'ی', 'ا', 'و', 'ه'].map(key); // in key form (ں -> ن)
|
||||||
|
export function stems(k: string) {
|
||||||
|
const out = new Set<string>();
|
||||||
|
for (const e of ENDINGS) {
|
||||||
|
if (!k.endsWith(e) || k.length - e.length < 2) continue;
|
||||||
|
const base = k.slice(0, -e.length);
|
||||||
|
for (const s of [base, base + 'ه', base + 'ا', base + 'نا']) if (s !== k) out.add(s);
|
||||||
|
}
|
||||||
|
return [...out];
|
||||||
|
}
|
||||||
|
|
||||||
|
// consonant skeleton as a regex: long vowels, heh and hamza optional (spellings add or drop them)
|
||||||
|
const SOFT = 'اوی\u0647ےءئ';
|
||||||
|
const esc = (c: string) => c.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
||||||
|
export function skeleton(k: string) {
|
||||||
|
const cs = [...k].filter((c) => !SOFT.includes(c));
|
||||||
|
if (cs.length < 2) return null;
|
||||||
|
const v = `[${SOFT}]*`;
|
||||||
|
return `^${v}${cs.map(esc).join(v)}${v}$`;
|
||||||
|
}
|
||||||
|
|
||||||
|
// shown titles without punctuation, dashes, spaces, tatweel or zero-width joiners
|
||||||
|
export const clean = (t: string) => t.replace(PUNCT, '').replace(/[\u0640\u200C\u200D]/g, '');
|
||||||
|
|
||||||
|
async function vowelVariants(k: string) {
|
||||||
|
const sk = skeleton(k);
|
||||||
|
if (!sk) return [];
|
||||||
|
const { rows } = await pool.query(
|
||||||
|
`SELECT key, (array_agg(title ORDER BY lang <> 'ur', lang <> 'fa', title))[1] AS title, array_agg(DISTINCT lang) AS langs
|
||||||
|
FROM wiktionary WHERE key ~ $1 AND key <> $2 GROUP BY key ORDER BY bool_or(lang = 'ur') DESC, similarity(key, $2) DESC LIMIT 10`, [sk, k]);
|
||||||
|
return rows.map((r) => ({ title: clean(r.title), langs: LANGS.filter((l) => r.langs.includes(l)) }));
|
||||||
|
}
|
||||||
|
|
||||||
|
async function similar(k: string) {
|
||||||
|
const st = stems(k), sk = skeleton(k), pre = [k, ...st].filter((s) => s.length >= 3).sort((a, b) => b.length - a.length)[0];
|
||||||
|
// one row per key: the Urdu spelling when there is one; Urdu entries first within each tier
|
||||||
|
const pick = `SELECT key, (array_agg(title ORDER BY lang <> 'ur', lang <> 'fa', title))[1] AS title, array_agg(DISTINCT lang) AS langs FROM wiktionary`;
|
||||||
|
const ur = `bool_or(lang = 'ur') DESC`;
|
||||||
|
const [a, b, c, d] = await Promise.all([
|
||||||
|
st.length ? pool.query(`${pick} WHERE key = ANY($1) GROUP BY key ORDER BY ${ur}`, [st]) : { rows: [] },
|
||||||
|
sk ? pool.query(`${pick} WHERE key ~ $1 AND key <> $2 GROUP BY key ORDER BY ${ur}, similarity(key, $2) DESC LIMIT 8`, [sk, k]) : { rows: [] },
|
||||||
|
pre ? pool.query(`${pick} WHERE key ~ $1 AND key <> $2 GROUP BY key ORDER BY ${ur}, length(key) LIMIT 8`, ['^' + [...pre].map(esc).join(''), k]) : { rows: [] },
|
||||||
|
pool.query(`${pick} WHERE key % $1 AND key <> $1 GROUP BY key ORDER BY ${ur}, similarity(key, $1) DESC LIMIT 8`, [k]),
|
||||||
|
]);
|
||||||
|
const seen = new Set<string>(), out: { title: string; langs: string[] }[] = [];
|
||||||
|
for (const r of [...a.rows, ...b.rows, ...c.rows, ...d.rows])
|
||||||
|
if (!seen.has(r.key) && seen.add(r.key)) out.push({ title: clean(r.title), langs: LANGS.filter((l) => r.langs.includes(l)) });
|
||||||
|
return out.slice(0, 12);
|
||||||
}
|
}
|
||||||
|
|||||||
@ -7,7 +7,7 @@
|
|||||||
import Fastify from 'fastify';
|
import Fastify from 'fastify';
|
||||||
import { pool } from './db.ts';
|
import { pool } from './db.ts';
|
||||||
import { likePatterns, normalise, terms } from './urdu.ts';
|
import { likePatterns, normalise, terms } from './urdu.ts';
|
||||||
import { lookup } from './dictionary.ts';
|
import { lookup, PUNCT } from './dictionary.ts';
|
||||||
|
|
||||||
const app = Fastify({ logger: { level: process.env.LOG_LEVEL ?? 'info' } });
|
const app = Fastify({ logger: { level: process.env.LOG_LEVEL ?? 'info' } });
|
||||||
const PAGE_SIZE = 20;
|
const PAGE_SIZE = 20;
|
||||||
@ -128,7 +128,7 @@ app.get<{ Querystring: { q?: string; poet?: string; page?: string } }>('/api/sea
|
|||||||
|
|
||||||
// one word in Arabic script (Urdu, Persian, Arabic), as selected by a reader
|
// one word in Arabic script (Urdu, Persian, Arabic), as selected by a reader
|
||||||
app.get<{ Querystring: { w?: string } }>('/api/word', async (req, reply) => {
|
app.get<{ Querystring: { w?: string } }>('/api/word', async (req, reply) => {
|
||||||
const w = (req.query.w ?? '').trim();
|
const w = (req.query.w ?? '').replace(PUNCT, ''); // commas, dots, dashes, spaces
|
||||||
if (!/^[\p{Script=Arabic}\p{M}\u200C]{1,40}$/u.test(w)) return reply.code(400).send({ error: 'one Urdu, Persian or Arabic word' });
|
if (!/^[\p{Script=Arabic}\p{M}\u200C]{1,40}$/u.test(w)) return reply.code(400).send({ error: 'one Urdu, Persian or Arabic word' });
|
||||||
return lookup(w);
|
return lookup(w);
|
||||||
});
|
});
|
||||||
|
|||||||
@ -70,6 +70,7 @@ CREATE TABLE IF NOT EXISTS wiktionary (
|
|||||||
);
|
);
|
||||||
CREATE INDEX IF NOT EXISTS wiktionary_key ON wiktionary(key);
|
CREATE INDEX IF NOT EXISTS wiktionary_key ON wiktionary(key);
|
||||||
CREATE INDEX IF NOT EXISTS wiktionary_title ON wiktionary(lang, source, title);
|
CREATE INDEX IF NOT EXISTS wiktionary_title ON wiktionary(lang, source, title);
|
||||||
|
CREATE INDEX IF NOT EXISTS wiktionary_key_trgm ON wiktionary USING gin (key gin_trgm_ops); -- similar words (regex, %)
|
||||||
CREATE TABLE IF NOT EXISTS ur_glosses ( -- English gloss -> Urdu word, for the pivot through English
|
CREATE TABLE IF NOT EXISTS ur_glosses ( -- English gloss -> Urdu word, for the pivot through English
|
||||||
gloss text NOT NULL,
|
gloss text NOT NULL,
|
||||||
word text NOT NULL
|
word text NOT NULL
|
||||||
|
|||||||
@ -136,44 +136,72 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
|
|||||||
dictBody.replaceChildren(el('h2', null, w), el('p', 'muted', 'تلاش جاری ہے…'));
|
dictBody.replaceChildren(el('h2', null, w), el('p', 'muted', 'تلاش جاری ہے…'));
|
||||||
let d = null;
|
let d = null;
|
||||||
// v: bump when the response format changes (responses are browser-cached for a day)
|
// v: bump when the response format changes (responses are browser-cached for a day)
|
||||||
try { const r = await fetch('/api/word?v=2&w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {}
|
try { const r = await fetch('/api/word?v=5&w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {}
|
||||||
if (w !== current) return; // a newer selection won
|
if (w !== current) return; // a newer selection won
|
||||||
|
const words = (title, items, into) => { // clickable word buttons
|
||||||
|
if (!items.length) return;
|
||||||
|
const box = el('div', 'similar');
|
||||||
|
for (const s of items) {
|
||||||
|
const b = el('button', null, s.title, { type: 'button', title: s.langs.map((c) => LANG[c]).join('، ') });
|
||||||
|
b.onclick = () => showWord(s.title);
|
||||||
|
box.append(b);
|
||||||
|
}
|
||||||
|
into.push(el('h3', null, title), box);
|
||||||
|
};
|
||||||
|
const list = (items, cls, attrs) => { const ol = el('ol', cls, null, attrs); items.forEach((x) => ol.append(el('li', null, x))); return ol; };
|
||||||
const out = [el('h2', null, w)];
|
const out = [el('h2', null, w)];
|
||||||
if (!d || !d.found) {
|
if (!d || !d.found) {
|
||||||
dictBody.replaceChildren(...out, el('p', 'muted', 'ویکی لغت میں یہ لفظ نہیں ملا۔'));
|
out.push(el('p', 'muted', 'ویکی لغت میں یہ لفظ نہیں ملا۔'));
|
||||||
|
words('ملتے جلتے الفاظ', [...(d?.similar ?? []), ...(d?.variants ?? [])], out);
|
||||||
|
dictBody.replaceChildren(...out);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
const list = (items, cls, attrs) => { const ol = el('ol', cls, null, attrs); items.forEach((x) => ol.append(el('li', null, x))); return ol; };
|
// one section per language, in order اردو، فارسی، عربی; each says where its meanings come from.
|
||||||
const equivalents = () => {
|
// A Persian or Arabic word is explained first by its Urdu translations, then English, then its own language.
|
||||||
const sec = el('section', 'dict-lang');
|
|
||||||
sec.append(el('h3', null, 'ہم معنی اردو الفاظ'), el('p', 'muted note', 'انگریزی کے واسطے سے'));
|
|
||||||
for (const e of d.equivalents) {
|
|
||||||
const p = el('p');
|
|
||||||
p.append(el('bdi', 'en', e.gloss), ': ' + e.urdu.join('، '));
|
|
||||||
sec.append(p);
|
|
||||||
}
|
|
||||||
return sec;
|
|
||||||
};
|
|
||||||
// one section per language, in order اردو، فارسی، عربی; each says where its meanings come from
|
|
||||||
for (const l of d.langs) {
|
for (const l of d.langs) {
|
||||||
const sec = el('section', 'dict-lang');
|
const sec = el('section', 'dict-lang');
|
||||||
sec.append(el('h3', null, LANG[l.code]));
|
sec.append(el('h3', null, LANG[l.code]));
|
||||||
const pr = el('p', 'pron');
|
// a pronunciation line: vowelled form, transliteration, IPA, audio
|
||||||
if (l.tr) pr.append(el('bdi', 'en', l.tr));
|
const pron = (r) => {
|
||||||
l.ipa.forEach((i) => pr.append(' ', el('bdi', 'ipa', i)));
|
const pr = el('p', 'pron');
|
||||||
if (l.audio && l.audio.startsWith('https://upload.wikimedia.org/')) {
|
if (r.form) pr.append(el('span', 'form', r.form), ' ');
|
||||||
const b = el('button', 'play', '🔊', { type: 'button', 'aria-label': 'تلفظ سنیں' });
|
if (r.tr) pr.append(el('bdi', 'en', r.tr));
|
||||||
b.onclick = () => new Audio(l.audio).play();
|
r.ipa.forEach((i) => pr.append(' ', el('bdi', 'ipa', i)));
|
||||||
pr.append(' ', b);
|
if (r.audio && r.audio.startsWith('https://upload.wikimedia.org/')) {
|
||||||
}
|
const b = el('button', 'play', '🔊', { type: 'button', 'aria-label': 'تلفظ سنیں' });
|
||||||
if (pr.childNodes.length) sec.append(pr);
|
b.onclick = () => new Audio(r.audio).play();
|
||||||
|
pr.append(' ', b);
|
||||||
|
}
|
||||||
|
return pr;
|
||||||
|
};
|
||||||
|
const urdu = () => {
|
||||||
|
if (!l.urdu?.length) return;
|
||||||
|
sec.append(el('p', 'pos', 'اردو میں · انگریزی ویکی لغت کے واسطے سے'));
|
||||||
|
for (const e of l.urdu) {
|
||||||
|
const p = el('p', 'via');
|
||||||
|
p.append(e.urdu.join('، '), ' ', el('bdi', 'en muted', `(${e.gloss})`));
|
||||||
|
sec.append(p);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// readings: the same spelling with different short vowels (ملک: مُلک، مَلِک، مِلک), each with its meanings
|
||||||
|
const english = () => {
|
||||||
|
if (l.readings.length > 1) sec.append(el('p', 'pos', `${l.readings.length} قراءتیں (حرکات کے فرق سے)`));
|
||||||
|
for (const r of l.readings) {
|
||||||
|
const box = el('div', l.readings.length > 1 ? 'reading' : null);
|
||||||
|
box.append(pron(r));
|
||||||
|
for (const s of r.senses) box.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos.charAt(0).toUpperCase() + s.pos.slice(1)] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' }));
|
||||||
|
sec.append(box);
|
||||||
|
}
|
||||||
|
if (l.translation.length) sec.append(el('p', 'pos', POS.Translation), list(l.translation, 'en', { dir: 'ltr', lang: 'en' }));
|
||||||
|
};
|
||||||
|
if (l.code !== 'ur') { urdu(); english(); }
|
||||||
if (l.meanings.length) {
|
if (l.meanings.length) {
|
||||||
const label = el('p', 'pos');
|
const label = el('p', 'pos');
|
||||||
label.append(`معانی: ${LANG[l.code]} ویکی لغت` + (l.origin ? ` · اصل: ${l.origin}` : '') + ' · ');
|
label.append(`معانی: ${LANG[l.code]} ویکی لغت` + (l.origin ? ` · اصل: ${l.origin}` : '') + ' · ');
|
||||||
if (l.source) label.append(el('a', null, 'ماخذ', { href: l.source, target: '_blank', rel: 'noopener' }));
|
if (l.source) label.append(el('a', null, 'ماخذ', { href: l.source, target: '_blank', rel: 'noopener' }));
|
||||||
sec.append(label, list(l.meanings, null, { lang: l.code }));
|
sec.append(label, list(l.meanings, null, { lang: l.code }));
|
||||||
}
|
}
|
||||||
for (const s of l.senses) sec.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos.charAt(0).toUpperCase() + s.pos.slice(1)] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' }));
|
if (l.code === 'ur') { english(); urdu(); }
|
||||||
if (l.synonyms.length) sec.append(el('p', 'pos', 'مترادفات'), el('p', null, l.synonyms.join('، ')));
|
if (l.synonyms.length) sec.append(el('p', 'pos', 'مترادفات'), el('p', null, l.synonyms.join('، ')));
|
||||||
if (l.ety) sec.append(el('p', 'pos', 'اشتقاق'), el('p', 'en ety', l.ety, { dir: 'ltr', lang: 'en' }));
|
if (l.ety) sec.append(el('p', 'pos', 'اشتقاق'), el('p', 'en ety', l.ety, { dir: 'ltr', lang: 'en' }));
|
||||||
if (l.en) {
|
if (l.en) {
|
||||||
@ -182,15 +210,14 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
|
|||||||
sec.append(p);
|
sec.append(p);
|
||||||
}
|
}
|
||||||
out.push(sec);
|
out.push(sec);
|
||||||
if (l.code === 'ur' && d.equivalents.length) out.push(equivalents());
|
|
||||||
}
|
}
|
||||||
if (d.equivalents.length && !d.langs.some((l) => l.code === 'ur')) out.splice(1, 0, equivalents());
|
words('ملتے جلتے الفاظ · دیگر حروفِ علت', d.variants ?? [], out); // same consonants, other long vowels
|
||||||
const src = el('p', 'src muted');
|
const src = el('p', 'src muted');
|
||||||
src.append('ویکی لغت (Wiktionary) · ');
|
src.append('ویکی لغت (Wiktionary) · ');
|
||||||
src.append('CC BY-SA');
|
src.append('CC BY-SA');
|
||||||
dictBody.replaceChildren(...out, src);
|
dictBody.replaceChildren(...out, src);
|
||||||
}
|
}
|
||||||
const TRIM = /^[\s"'«»()\[\]،۔؟؛!:.,]+|[\s"'«»()\[\]،۔؟؛!:.,]+$/g;
|
const TRIM = /["'«»()\[\]،۔؟؛!:.,\-–—_]+|^\s+|\s+$/g; // punctuation and dashes anywhere, outer spaces
|
||||||
let selTimer;
|
let selTimer;
|
||||||
document.addEventListener('selectionchange', () => {
|
document.addEventListener('selectionchange', () => {
|
||||||
clearTimeout(selTimer);
|
clearTimeout(selTimer);
|
||||||
@ -198,6 +225,7 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
|
|||||||
const sel = getSelection();
|
const sel = getSelection();
|
||||||
if (!sel || sel.isCollapsed || !main.contains(sel.anchorNode)) return;
|
if (!sel || sel.isCollapsed || !main.contains(sel.anchorNode)) return;
|
||||||
const w = sel.toString().replace(TRIM, '').replace(/\u0614/g, ''); // drop the takhallus sign
|
const w = sel.toString().replace(TRIM, '').replace(/\u0614/g, ''); // drop the takhallus sign
|
||||||
|
if (/\s/.test(w)) return; // one word only
|
||||||
if (/^[\p{Script=Arabic}\p{M}\u200C]{1,40}$/u.test(w)) showWord(w);
|
if (/^[\p{Script=Arabic}\p{M}\u200C]{1,40}$/u.test(w)) showWord(w);
|
||||||
}, 350);
|
}, 350);
|
||||||
});
|
});
|
||||||
|
|||||||
@ -145,9 +145,16 @@ h1 + .muted { text-align: center; margin-top: 0; }
|
|||||||
.dict .pron { margin: 2px 0; }
|
.dict .pron { margin: 2px 0; }
|
||||||
.dict .pos { font-size: .8rem; color: var(--muted); margin: 8px 0 0; }
|
.dict .pos { font-size: .8rem; color: var(--muted); margin: 8px 0 0; }
|
||||||
.dict .note { font-size: .8rem; margin: 0; }
|
.dict .note { font-size: .8rem; margin: 0; }
|
||||||
|
.dict .via { margin: 2px 0; }
|
||||||
|
.dict .reading { border-inline-start: 2px solid var(--gold-light); padding-inline-start: 10px; margin: 8px 0; }
|
||||||
|
.dict .form { font-size: 1.15rem; color: var(--lapis); }
|
||||||
|
.dict .via .en { font-size: .78rem; }
|
||||||
.dict .src { font-size: .78rem; margin-top: 18px; }
|
.dict .src { font-size: .78rem; margin-top: 18px; }
|
||||||
.dict .ety { text-align: left; color: var(--muted); font-size: .8rem; }
|
.dict .ety { text-align: left; color: var(--muted); font-size: .8rem; }
|
||||||
.dict-close { position: sticky; top: 0; float: left; border: 0; background: none; color: var(--muted); font-size: 1rem; cursor: pointer; }
|
.dict-close { position: sticky; top: 0; float: left; border: 0; background: none; color: var(--muted); font-size: 1rem; cursor: pointer; }
|
||||||
|
.dict .similar { display: flex; flex-wrap: wrap; gap: 6px; }
|
||||||
|
.dict .similar button { font: inherit; border: 1px solid var(--border); background: var(--inner); color: var(--ink); border-radius: 999px; padding: 0 12px; cursor: pointer; }
|
||||||
|
.dict .similar button:hover { border-color: var(--brand); color: var(--brand); }
|
||||||
.dict .play { border: 1px solid var(--border); background: var(--inner); border-radius: 8px; cursor: pointer; padding: 0 6px; }
|
.dict .play { border: 1px solid var(--border); background: var(--inner); border-radius: 8px; cursor: pointer; padding: 0 6px; }
|
||||||
/* wide screens: the page moves over instead of being covered */
|
/* wide screens: the page moves over instead of being covered */
|
||||||
@media (min-width: 1100px) { body.dict-open { padding-left: min(340px, 92vw); } }
|
@media (min-width: 1100px) { body.dict-open { padding-left: min(340px, 92vw); } }
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user