Word dictionary: full Wiktionary data for Urdu, Persian and Arabic in PostgreSQL, kept in sync
- en.wiktionary entries via kaikki.org (meanings, IPA, transliteration, audio, etymology, synonyms) - ur./fa./ar.wiktionary dumps for meanings in each language (incl. more Urdu entry formats) - English->Urdu pivot table from Urdu entries' glosses and ur.wiktionary's English entries - dict-sync.ts: re-imports sources when upstream publishes new files, applies daily recent changes; run from deploy/sync.sh - lookups now read the database (2-7 ms) instead of calling Wikimedia live Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
ef618094d5
commit
7fcc0ee7fa
@ -42,6 +42,8 @@ The import upserts, so re-running it after a divan-data sync applies the changes
|
|||||||
15 3 * * * DATABASE_URL=postgres://divan:...@localhost:5432/divan /opt/divan/deploy/sync.sh >> /var/log/divan-sync.log 2>&1
|
15 3 * * * DATABASE_URL=postgres://divan:...@localhost:5432/divan /opt/divan/deploy/sync.sh >> /var/log/divan-sync.log 2>&1
|
||||||
```
|
```
|
||||||
|
|
||||||
|
The same script keeps the word dictionary current (`npm run dict-sync` in `api/`): the full Wiktionary data for Urdu, Persian and Arabic (English Wiktionary via [kaikki.org](https://kaikki.org), and the Urdu, Persian and Arabic Wiktionary dumps) is re-imported when upstream publishes new files, and each day's Wiktionary edits are applied from recent changes. The first run downloads about 700 MB.
|
||||||
|
|
||||||
Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DIVAN_APP_DIR` (default: this repo), `DIVAN_DATA_PUSH=1` to also commit and push data changes (needs git push access). Needs git, python3 and Node 24+.
|
Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DIVAN_APP_DIR` (default: this repo), `DIVAN_DATA_PUSH=1` to also commit and push data changes (needs git push access). Needs git, python3 and Node 24+.
|
||||||
|
|
||||||
## API
|
## API
|
||||||
@ -51,7 +53,7 @@ Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DI
|
|||||||
| `GET /api/poets` | all poets |
|
| `GET /api/poets` | all poets |
|
||||||
| `GET /api/page?url=/p238/...` | the poet, category or poem at a site URL (breadcrumbs, children, verses, prev/next) |
|
| `GET /api/page?url=/p238/...` | the poet, category or poem at a site URL (breadcrumbs, children, verses, prev/next) |
|
||||||
| `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised; exact phrase first; each with the best-matching couplet or paragraph (`snippet`) |
|
| `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised; exact phrase first; each with the best-matching couplet or paragraph (`snippet`) |
|
||||||
| `GET /api/word?w=` | one word's meanings and pronunciation from Wiktionary (Urdu, Persian, Arabic in that order; English meanings; Urdu equivalents via English when Urdu Wiktionary has none), cached 30 days |
|
| `GET /api/word?w=` | one word's meanings and pronunciation from the local Wiktionary data (Urdu, Persian, Arabic in that order; English meanings; Urdu equivalents via English when Urdu Wiktionary has none) |
|
||||||
| `GET /health` | database check |
|
| `GET /health` | database check |
|
||||||
|
|
||||||
Search normalises both stored text and queries: Arabic ي/ك/ه → Urdu ی/ک/ہ, ۂ/ۓ, diacritics and the Urdu full stop removed; do-chashmi ھ stays distinct.
|
Search normalises both stored text and queries: Arabic ي/ك/ه → Urdu ی/ک/ہ, ۂ/ۓ, diacritics and the Urdu full stop removed; do-chashmi ھ stays distinct.
|
||||||
|
|||||||
@ -9,7 +9,8 @@
|
|||||||
"scripts": {
|
"scripts": {
|
||||||
"start": "node src/server.ts",
|
"start": "node src/server.ts",
|
||||||
"import": "node src/import.ts",
|
"import": "node src/import.ts",
|
||||||
"test": "node --test src/*.test.ts"
|
"test": "node --test src/*.test.ts",
|
||||||
|
"dict-sync": "node src/dict-sync.ts"
|
||||||
},
|
},
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"fastify": "^5.12.5",
|
"fastify": "^5.12.5",
|
||||||
|
|||||||
10
api/src/dict-sync.ts
Normal file
10
api/src/dict-sync.ts
Normal file
@ -0,0 +1,10 @@
|
|||||||
|
// Import / sync the Wiktionary data (see dictionary.ts). First run imports everything (about 1 GB of downloads);
|
||||||
|
// later runs re-import only sources that changed upstream and apply the day's Wiktionary edits.
|
||||||
|
// node src/dict-sync.ts
|
||||||
|
import { readFile } from 'node:fs/promises';
|
||||||
|
import { pool } from './db.ts';
|
||||||
|
import { sync } from './dictionary.ts';
|
||||||
|
|
||||||
|
await pool.query(await readFile(new URL('../../db/schema.sql', import.meta.url), 'utf8'));
|
||||||
|
await sync((line) => console.log(`${new Date().toISOString()} ${line}`));
|
||||||
|
await pool.end();
|
||||||
@ -1,34 +1,67 @@
|
|||||||
import { test } from 'node:test';
|
import { test } from 'node:test';
|
||||||
import assert from 'node:assert/strict';
|
import assert from 'node:assert/strict';
|
||||||
import { forms, urduEntry, pronunciations } from './dictionary.ts';
|
import { key, kaikkiEntry, glossKeys, urduEntry, definitions, dumpPages } from './dictionary.ts';
|
||||||
|
|
||||||
test('spelling forms: diacritics dropped, final noon ghunna tried as noon', () => {
|
test('lookup key: spelling variants across Urdu, Persian and Arabic meet', () => {
|
||||||
assert.deepEqual(forms('ناداں'), ['ناداں', 'نادان']);
|
const same = (a: string, b: string) => assert.equal(key(a), key(b), `${a} vs ${b}`);
|
||||||
assert.deepEqual(forms('وِصال'), ['وِصال', 'وصال']);
|
same('ناداں', 'نادان');
|
||||||
|
same('قِسْمَت', 'قسمت');
|
||||||
|
same('نگاہ', 'نگاه');
|
||||||
|
same('معنی', 'معنى');
|
||||||
|
same('كتاب', 'کتاب');
|
||||||
|
assert.notEqual(key('کھ'), key('کہ'), 'do-chashmi heh stays distinct');
|
||||||
|
assert.notEqual(key('دیوانے'), key('دیوانی'), 'bari ye stays distinct');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('kaikki entry: glosses, IPA, audio, romanisation, form-of', () => {
|
||||||
|
const e = kaikkiEntry({
|
||||||
|
word: 'قسمت', pos: 'noun', etymology_text: 'From Arabic',
|
||||||
|
senses: [{ glosses: ['fate, destiny'] }, { glosses: ['division'] }, { links: [] }],
|
||||||
|
sounds: [{ ipa: '/qɪs.mət̪/' }, { audio: 'x.wav', mp3_url: 'https://upload.wikimedia.org/x.mp3' }, { mp3_url: 'https://evil.example/x.mp3' }],
|
||||||
|
forms: [{ form: 'قِسْمَت', tags: ['canonical'] }, { form: 'qismat', tags: ['romanization'] }],
|
||||||
|
});
|
||||||
|
assert.deepEqual(e, { pos: 'noun', glosses: ['fate, destiny', 'division'], formOf: false, ipa: ['/qɪs.mət̪/'],
|
||||||
|
audio: 'https://upload.wikimedia.org/x.mp3', tr: 'qismat', ety: 'From Arabic', synonyms: [] });
|
||||||
|
assert.equal(kaikkiEntry({ word: 'x', pos: 'noun', senses: [{ glosses: ['plural of y'], tags: ['form-of'] }] }).formOf, true);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('pivot keys from English glosses', () => {
|
||||||
|
assert.deepEqual(glossKeys(['fate, destiny', 'to love (someone)', 'a very long description of something that is not a key']), ['fate', 'destiny', 'love']);
|
||||||
});
|
});
|
||||||
|
|
||||||
test('ur.wiktionary entry: meanings and origin', () => {
|
test('ur.wiktionary entry: meanings and origin', () => {
|
||||||
const wt = "وِصال {وِصال} ([[عربی]])\n\nاسم [[نکرہ]]\n\n==معانی==\n\n1. بھیٹ، ملاقات، [[عاشق و معشوق]] کی محبت۔\n\n==مرکبات==\n\nوِصال کا دِن";
|
const wt = 'وِصال {وِصال} ([[عربی]])\n\nاسم [[نکرہ]]\n\n==معانی==\n\n1. بھیٹ، ملاقات، [[عاشق و معشوق]] کی محبت۔\n\n==مرکبات==\n\nوِصال کا دِن';
|
||||||
assert.deepEqual(urduEntry(wt), { meanings: ['بھیٹ، ملاقات، عاشق و معشوق کی محبت۔'], origin: 'عربی' });
|
assert.deepEqual(urduEntry(wt), { defs: ['بھیٹ، ملاقات، عاشق و معشوق کی محبت۔'], origin: 'عربی', links: [] });
|
||||||
});
|
});
|
||||||
|
|
||||||
test('en.wiktionary HTML: IPA, transliteration and audio per language', () => {
|
test('fa./ar.wiktionary definitions: skip etymology, Arabic pointers', () => {
|
||||||
const html = '<h2 data-mw-wikitext="" id="Persian">Persian</h2><span class="IPA nowrap">/qis.ˈmat/</span><span class="IPA">-at</span>' +
|
|
||||||
'<span lang="fa-Latn" class="headword-tr manual-tr tr Latn" dir="ltr">qismat</span><audio><source src="//upload.wikimedia.org/a/b.ogg"></audio>' +
|
|
||||||
'<h2 id="Urdu">Urdu</h2><span class="IPA nowrap">/qɪs.mət̪/</span><span lang="ur-Latn" class="headword-tr tr Latn">qismat</span>';
|
|
||||||
assert.deepEqual(pronunciations(html), {
|
|
||||||
fa: { ipa: ['/qis.ˈmat/'], tr: 'qismat', audio: 'https://upload.wikimedia.org/a/b.ogg' },
|
|
||||||
ur: { ipa: ['/qɪs.mət̪/'], tr: 'qismat', audio: null },
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
test('fa./ar.wiktionary definitions: skip etymology, follow Arabic pointers', async () => {
|
|
||||||
const { definitions, formsIn } = await import('./dictionary.ts');
|
|
||||||
const fa = '==فارسی==\n===ریشهشناسی===\n* [[عربی]]\n# قسمة\n===اسم===\n#بهره، نصیب.\n#:example\n#سرنوشت، تقدیر.\n====برگردانها====\n# x y z';
|
const fa = '==فارسی==\n===ریشهشناسی===\n* [[عربی]]\n# قسمة\n===اسم===\n#بهره، نصیب.\n#:example\n#سرنوشت، تقدیر.\n====برگردانها====\n# x y z';
|
||||||
assert.deepEqual(definitions(fa, 'fa').defs, ['بهره، نصیب.', 'سرنوشت، تقدیر.']);
|
assert.deepEqual(definitions(fa, 'fa').defs, ['بهره، نصیب.', 'سرنوشت، تقدیر.']);
|
||||||
const ar = '== {{اللغة|عربية}} ==\nهل تقصد:\n* [[عِشْق]]\n* [[عَشَقَ]]\n----\n== {{اللغة|أردية}} ==\n# [[x]]';
|
const ar = '== {{اللغة|عربية}} ==\nهل تقصد:\n* [[عِشْق]]\n* [[عَشَقَ]]\n----\n== {{اللغة|أردية}} ==\n# [[x]]';
|
||||||
assert.deepEqual(definitions(ar, 'ar'), { defs: [], links: ['عِشْق', 'عَشَقَ'] });
|
assert.deepEqual(definitions(ar, 'ar'), { defs: [], origin: null, links: ['عِشْق', 'عَشَقَ'] });
|
||||||
assert.deepEqual(definitions('# {{مصدر|عَشِقَ}}.\n# فرط الحب.', 'ar').defs, ['فرط الحب.']);
|
assert.deepEqual(definitions('# {{مصدر|عَشِقَ}}.\n# فرط الحب.', 'ar').defs, ['فرط الحب.']);
|
||||||
assert.deepEqual(formsIn('fa', 'نگاہ'), ['نگاه']);
|
});
|
||||||
assert.deepEqual(formsIn('ar', 'معنی'), ['معني', 'معنى']);
|
|
||||||
|
test('dump pages: articles only, entities decoded, redirects skipped', () => {
|
||||||
|
const xml = '<page><title>عشق</title><ns>0</ns><revision><text bytes="9">a & b <x></text></revision></page>' +
|
||||||
|
'<page><title>Talk:x</title><ns>1</ns><text>t</text></page><page><title>r</title><ns>0</ns><redirect title="y" /><text>#R</text></page>';
|
||||||
|
assert.deepEqual([...dumpPages(xml)], [{ title: 'عشق', text: 'a & b <x>' }]);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('etymology cut never leaves half a surrogate pair', () => {
|
||||||
|
const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'a'.repeat(299) + '𑀭𑀭' }).ety!;
|
||||||
|
assert.ok(ety.isWellFormed());
|
||||||
|
});
|
||||||
|
|
||||||
|
test('ur.wiktionary: English entries give Urdu words; Urdu entries fall back to # lines or prose', async () => {
|
||||||
|
const { englishToUrdu, isEnglish } = await import('./dictionary.ts');
|
||||||
|
assert.ok(isEnglish('north') && !isEnglish('شمال'));
|
||||||
|
assert.deepEqual(englishToUrdu('==انگریزی==\n===صفت===\n{{en-adjective}}\n# [[شمالی]]۔\n# [[شمال]] کی [[جانب]]۔\n#: [[x]]'), ['شمالی', 'شمال', 'جانب']);
|
||||||
|
assert.deepEqual(urduEntry('===اسم===\n# [[بہار]] کا [[موسم]]۔').defs, ['بہار کا موسم۔']);
|
||||||
|
assert.deepEqual(urduEntry("'''پھوڑی''' اس چادر کو کہتے ہیں جس پر بیٹھتے ہیں۔").defs, ['پھوڑی اس چادر کو کہتے ہیں جس پر بیٹھتے ہیں۔']);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('etymology drops the "Etymology tree" summary', () => {
|
||||||
|
const ety = kaikkiEntry({ word: 'x', pos: 'noun', senses: [], etymology_text: 'Etymology tree Arabic قَسَمَ (qasama)bor. Urdu قِسْمَت Borrowed from Classical Persian قِسْمَت (qismat).' }).ety;
|
||||||
|
assert.equal(ety, 'Borrowed from Classical Persian قِسْمَت (qismat).');
|
||||||
});
|
});
|
||||||
|
|||||||
@ -1,48 +1,89 @@
|
|||||||
// Word lookup for the reading sidebar, from Wiktionary (CC BY-SA). For Urdu, Persian and Arabic, in that order:
|
// Word dictionary for the reading sidebar: the full Wiktionary data for Urdu, Persian and Arabic, in PostgreSQL.
|
||||||
// - meanings in the language itself from its own Wiktionary (ur., fa., ar.wiktionary)
|
// source 'en': en.wiktionary entries (English meanings, IPA, transliteration, audio, etymology, synonyms),
|
||||||
// - English meanings and pronunciation (IPA, transliteration, audio) from en.wiktionary
|
// from the kaikki.org Wiktextract extracts (weekly)
|
||||||
// With no Urdu meanings, Urdu equivalents are pivoted through English (the English glosses' translation tables).
|
// source 'own': each language's own Wiktionary (ur., fa., ar.wiktionary): meanings in that language,
|
||||||
// Found words are cached in PostgreSQL (dictionary table) for 30 days.
|
// from the Wikimedia dumps (twice a month) plus recent changes (daily)
|
||||||
|
// Urdu equivalents for words with no Urdu meaning are pivoted through English: ur_glosses maps English words to
|
||||||
|
// Urdu words, from the English glosses of en.wiktionary's Urdu entries and ur.wiktionary's English entries. Data: CC BY-SA, Wiktionary contributors.
|
||||||
|
import { spawn } from 'node:child_process';
|
||||||
|
import { createInterface } from 'node:readline';
|
||||||
|
import { Readable } from 'node:stream';
|
||||||
import { pool } from './db.ts';
|
import { pool } from './db.ts';
|
||||||
|
|
||||||
|
export const LANGS = ['ur', 'fa', 'ar'] as const;
|
||||||
|
const NAME = { ur: 'Urdu', fa: 'Persian', ar: 'Arabic' } as const;
|
||||||
const UA = 'Divan/2 (https://github.com/anas-rashid/divan)';
|
const UA = 'Divan/2 (https://github.com/anas-rashid/divan)';
|
||||||
const LANGS = { ur: 'Urdu', fa: 'Persian', ar: 'Arabic' } as const;
|
const KAIKKI = (l: string) => `https://kaikki.org/dictionary/${NAME[l as 'ur']}/kaikki.org-dictionary-${NAME[l as 'ur']}.jsonl`;
|
||||||
const DIAC = /[ً-ْٰٔٗ٘]/g;
|
const DUMP = (l: string) => `https://dumps.wikimedia.org/${l}wiktionary/latest/${l}wiktionary-latest-pages-articles.xml.bz2`;
|
||||||
const plain = (html: string) =>
|
const MARKS = /[ؐ-ًؚ-ٰٟۖ-ۭـ]/g;
|
||||||
html.replace(/<[^>]+>/g, '').replace(/ /g, ' ').replace(/&/g, '&').replace(/&#\d+;/g, '').replace(/\s+/g, ' ').trim();
|
|
||||||
|
// spelling-insensitive key shared by Urdu, Persian and Arabic: no diacritics/tatweel, one heh, one yeh, one kaf,
|
||||||
|
// plain alef, noon ghunna as noon (ناداں = نادان, قسمت = قِسْمَت, نگاہ = نگاه, معنی = معنى). ے and ھ stay distinct.
|
||||||
|
export const key = (w: string) =>
|
||||||
|
w.normalize('NFC').replace(MARKS, '')
|
||||||
|
.replace(/[ہۂۃةه]/g, 'ه').replace(/[يىی]/g, 'ی')
|
||||||
|
.replace(/ك/g, 'ک').replace(/[أإٱ]/g, 'ا').replace(/ں/g, 'ن').trim();
|
||||||
|
|
||||||
const unlink = (wt: string) => wt.replace(/\[\[(?:[^\]|]*\|)?([^\]]*)\]\]/g, '$1').replace(/'''?/g, '').trim();
|
const unlink = (wt: string) => wt.replace(/\[\[(?:[^\]|]*\|)?([^\]]*)\]\]/g, '$1').replace(/'''?/g, '').trim();
|
||||||
|
const upload = (u?: string) => (u && u.startsWith('https://upload.wikimedia.org/') ? u : null);
|
||||||
|
|
||||||
async function get(url: string, json = true): Promise<any> {
|
// ---- parsing ----
|
||||||
const res = await fetch(url, { headers: { 'user-agent': UA }, signal: AbortSignal.timeout(8000) }).catch(() => null);
|
|
||||||
if (!res?.ok) return null;
|
|
||||||
return json ? res.json() : res.text();
|
|
||||||
}
|
|
||||||
const rest = (path: string, title: string) => `https://en.wiktionary.org/api/rest_v1/page/${path}/${encodeURIComponent(title)}`;
|
|
||||||
const wikitext = async (host: string, title: string): Promise<string> => {
|
|
||||||
const d = await get(`https://${host}/w/api.php?action=parse&page=${encodeURIComponent(title)}&prop=wikitext&format=json&formatversion=2&redirects=1`);
|
|
||||||
return d?.parse?.wikitext ?? '';
|
|
||||||
};
|
|
||||||
|
|
||||||
// spelling forms to try: as written, without diacritics, final noon ghunna as noon (ناداں -> نادان)
|
// a kaikki.org (Wiktextract) line -> the fields the sidebar shows
|
||||||
export const forms = (w: string) => {
|
export function kaikkiEntry(d: any) {
|
||||||
const bare = w.replace(DIAC, '').replace(/ۂ/g, 'ہ');
|
const senses = (d.senses ?? []).filter((s: any) => s.glosses?.length);
|
||||||
return [...new Set([w, bare, bare.replace(/ں$/, 'ن')])];
|
const sounds = d.sounds ?? [];
|
||||||
};
|
return {
|
||||||
|
pos: d.pos as string,
|
||||||
// the word in Persian / Arabic spelling (ہ -> ه, ی -> ي, ک -> ك …); Arabic also tries final alef maqsura
|
glosses: senses.map((s: any) => s.glosses.join('; ')).slice(0, 6) as string[],
|
||||||
const SPELL: Record<string, [RegExp, string][]> = {
|
formOf: senses.length > 0 && senses.every((s: any) => s.form_of || s.tags?.includes('form-of')),
|
||||||
fa: [[/[\u06C1\u06C2\u06BE]/g, '\u0647'], [/[\u06D2\u064A]/g, '\u06CC'], [/\u0643/g, '\u06A9']],
|
ipa: [...new Set(sounds.map((s: any) => s.ipa).filter(Boolean))].slice(0, 2) as string[],
|
||||||
ar: [[/[\u06C1\u06BE]/g, '\u0647'], [/\u06C2/g, '\u0629'], [/[\u06CC\u06D2]/g, '\u064A'], [/\u06A9/g, '\u0643']],
|
audio: sounds.map((s: any) => upload(s.mp3_url) ?? upload(s.ogg_url)).find(Boolean) ?? null,
|
||||||
};
|
tr: d.forms?.find((f: any) => f.tags?.includes('romanization'))?.form ?? null,
|
||||||
export function formsIn(code: string, w: string) {
|
ety: d.etymology_text ? etymology(String(d.etymology_text)) : null,
|
||||||
const base = forms(w);
|
synonyms: (d.synonyms ?? []).map((s: any) => s.word).filter(Boolean).slice(0, 8) as string[],
|
||||||
if (!SPELL[code]) return base;
|
};
|
||||||
const conv = base.map((f) => SPELL[code].reduce((s, [re, to]) => s.replace(re, to), f));
|
|
||||||
return [...new Set([...conv, ...(code === 'ar' ? conv.map((f) => f.replace(/\u064A$/, '\u0649')) : [])])];
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// fa./ar.wiktionary: "# definition" lines outside etymology/translation sections; on ar.wiktionary only
|
// etymology text without Wiktionary's "Etymology tree …" diagram summary; cut at 300 characters, never
|
||||||
// the Arabic section, and undiacritised pages that just point to entries ("* [[عِشْق]]") give those links
|
// leaving half a surrogate pair
|
||||||
|
const etymology = (t: string) =>
|
||||||
|
(t.startsWith('Etymology tree') ? t.slice(Math.max(0, t.search(/\b(Borrowed|Inherited|From|Learned|Derived|Semi-learned|Calque|Compound)\b/))) : t)
|
||||||
|
.slice(0, 300).toWellFormed();
|
||||||
|
|
||||||
|
// English glosses of an Urdu entry as pivot keys: "fate, destiny" -> fate, destiny (short, lowercase)
|
||||||
|
export const glossKeys = (glosses: string[]) => [...new Set(glosses.flatMap((g) => g.replace(/\([^)]*\)/g, '').split(/[,;]/))
|
||||||
|
.map((s) => s.trim().toLowerCase().replace(/^(to|a|an|the) /, '')).filter((s) => /^[a-z][a-z' -]{1,30}$/.test(s) && s.split(' ').length <= 3))];
|
||||||
|
|
||||||
|
// ur.wiktionary Urdu entries: numbered lines under ==معانی==, else "# " lines, else the opening prose;
|
||||||
|
// origin from "(عربی)" on the first line
|
||||||
|
export function urduEntry(wt: string) {
|
||||||
|
const m = wt.split(/==\s*معانی\s*==/)[1]?.split(/\n==/)[0] ?? '';
|
||||||
|
let defs = m.split('\n').map((l) => l.match(/^\s*(?:\d+[.)-]|#)\s*(.+)/)?.[1]).filter(Boolean).map((l) => unlink(l!));
|
||||||
|
if (!defs.length) defs = definitions(wt, 'ur').defs;
|
||||||
|
if (!defs.length) {
|
||||||
|
const prose = wt.split('\n').find((l) => l.trim().startsWith("'''") && l.length > 20);
|
||||||
|
if (prose) defs = [unlink(stripTemplates(prose)).slice(0, 300)];
|
||||||
|
}
|
||||||
|
const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null;
|
||||||
|
const english = wt.match(/^\s*انگریزی\s*:\s*([A-Za-z].+)$/m)?.[1].trim() ?? null; // "== تراجم ==" pages
|
||||||
|
return { defs: defs.slice(0, 6), origin, links: [] as string[], ...(english && { english }) };
|
||||||
|
}
|
||||||
|
|
||||||
|
// ur.wiktionary English entries ("north": "# [[شمالی]]۔"): the Urdu words, for the pivot through English
|
||||||
|
export const isEnglish = (title: string) => /^[A-Za-z][A-Za-z' -]*$/.test(title);
|
||||||
|
export function englishToUrdu(wt: string) {
|
||||||
|
const lines = wt.split('\n').filter((l) => /^#(?![:*])/.test(l)).join(' ');
|
||||||
|
return [...new Set([...lines.matchAll(/\[\[(?:[^\]|]*\|)?([^\]]+)\]\]/g)].map((m) => m[1].trim())
|
||||||
|
.filter((w) => /^[\p{Script=Arabic}\s]+$/u.test(w)))].slice(0, 10);
|
||||||
|
}
|
||||||
|
const stripTemplates = (t: string) => {
|
||||||
|
while (/\{\{[^{}]*\}\}/.test(t)) t = t.replace(/\{\{[^{}]*\}\}/g, '');
|
||||||
|
return t;
|
||||||
|
};
|
||||||
|
|
||||||
|
// fa./ar.wiktionary: "# definition" lines outside etymology/translation sections; on ar.wiktionary only the
|
||||||
|
// Arabic section, and undiacritised pages that just point to entries ("* [[عِشْق]]") give those links
|
||||||
export function definitions(wt: string, code: string) {
|
export function definitions(wt: string, code: string) {
|
||||||
if (code === 'ar' && wt.includes('{{اللغة|')) wt = wt.split(/==\s*\{\{اللغة\|/).find((x) => x.startsWith('عربية')) ?? '';
|
if (code === 'ar' && wt.includes('{{اللغة|')) wt = wt.split(/==\s*\{\{اللغة\|/).find((x) => x.startsWith('عربية')) ?? '';
|
||||||
const defs: string[] = [], links: string[] = [];
|
const defs: string[] = [], links: string[] = [];
|
||||||
@ -55,102 +96,214 @@ export function definitions(wt: string, code: string) {
|
|||||||
if (only) { links.push(only[1]); continue; }
|
if (only) { links.push(only[1]); continue; }
|
||||||
const m = line.match(/^#(?![:*])\s*(.+)/);
|
const m = line.match(/^#(?![:*])\s*(.+)/);
|
||||||
if (!m) continue;
|
if (!m) continue;
|
||||||
let t = m[1];
|
const t = unlink(stripTemplates(m[1])).replace(/^[\s.،:-]+|\s+$/g, '');
|
||||||
while (/\{\{[^{}]*\}\}/.test(t)) t = t.replace(/\{\{[^{}]*\}\}/g, '');
|
|
||||||
t = unlink(t).replace(/^[\s.،:-]+|[\s]+$/g, '');
|
|
||||||
if (t.length >= 3 && !/^-+$/.test(t)) defs.push(t);
|
if (t.length >= 3 && !/^-+$/.test(t)) defs.push(t);
|
||||||
}
|
}
|
||||||
return { defs: defs.slice(0, 5), links };
|
return { defs: defs.slice(0, 6), origin: null as string | null, links: links.slice(0, 5) };
|
||||||
}
|
}
|
||||||
|
|
||||||
// meanings from the language's own Wiktionary: the first spelling with an entry (following one pointer hop)
|
export const ownEntry = (code: string, wt: string) => (code === 'ur' ? urduEntry(wt) : definitions(wt, code));
|
||||||
async function native(code: string, word: string) {
|
|
||||||
const host = `${code}.wiktionary.org`;
|
// pages of a MediaWiki XML dump (articles only, no redirects)
|
||||||
for (const f of formsIn(code, word)) {
|
export function* dumpPages(xml: string) {
|
||||||
const wt = await wikitext(host, f);
|
for (const m of xml.matchAll(/<page>([\s\S]*?)<\/page>/g)) {
|
||||||
if (!wt) continue;
|
const p = m[1];
|
||||||
if (code === 'ur') {
|
if (!/<ns>0<\/ns>/.test(p) || /<redirect /.test(p)) continue;
|
||||||
const e = urduEntry(wt);
|
const title = p.match(/<title>([^<]*)<\/title>/)?.[1];
|
||||||
if (e.meanings.length) return { defs: e.meanings, origin: e.origin, url: `https://${host}/wiki/${encodeURIComponent(f)}` };
|
const text = p.match(/<text[^>]*>([\s\S]*?)<\/text>/)?.[1];
|
||||||
|
if (title && text) yield { title: xmlText(title), text: xmlText(text) };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const xmlText = (s: string) => s.replace(/</g, '<').replace(/>/g, '>').replace(/"/g, '"').replace(/'/g, "'").replace(/&/g, '&');
|
||||||
|
|
||||||
|
// ---- import and sync ----
|
||||||
|
|
||||||
|
const meta = async (k: string) => (await pool.query('SELECT value FROM dict_meta WHERE name = $1', [k])).rows[0]?.value ?? null;
|
||||||
|
const setMeta = (k: string, v: string) =>
|
||||||
|
pool.query('INSERT INTO dict_meta (name, value) VALUES ($1, $2) ON CONFLICT (name) DO UPDATE SET value = $2', [k, v]);
|
||||||
|
|
||||||
|
async function insertRows(client: any, rows: unknown[][]) {
|
||||||
|
for (let i = 0; i < rows.length; i += 500) {
|
||||||
|
const chunk = rows.slice(i, i + 500), values: unknown[] = [];
|
||||||
|
const sql = chunk.map((r, j) => { values.push(...r); const o = j * 5; return `($${o + 1},$${o + 2},$${o + 3},$${o + 4},$${o + 5})`; });
|
||||||
|
await client.query(`INSERT INTO wiktionary (lang, source, title, key, data) VALUES ${sql.join(',')}`, values);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// replace one source's rows in a transaction (readers see the old data until it commits)
|
||||||
|
async function replace(lang: string, source: string, fill: (add: (title: string, data: object) => Promise<void>, client: any) => Promise<void>) {
|
||||||
|
const client = await pool.connect();
|
||||||
|
let n = 0;
|
||||||
|
try {
|
||||||
|
await client.query('BEGIN');
|
||||||
|
await client.query('DELETE FROM wiktionary WHERE lang = $1 AND source = $2', [lang, source]);
|
||||||
|
if (lang === 'ur') await client.query('DELETE FROM ur_glosses WHERE source = $1', [source]);
|
||||||
|
let batch: unknown[][] = [];
|
||||||
|
await fill(async (title, data) => {
|
||||||
|
batch.push([lang, source, title, key(title), data]); n++;
|
||||||
|
if (batch.length >= 2000) { await insertRows(client, batch); batch = []; }
|
||||||
|
}, client);
|
||||||
|
await insertRows(client, batch);
|
||||||
|
await client.query('COMMIT');
|
||||||
|
} catch (e) {
|
||||||
|
await client.query('ROLLBACK');
|
||||||
|
throw e;
|
||||||
|
} finally {
|
||||||
|
client.release();
|
||||||
|
}
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
|
async function lastModified(url: string) {
|
||||||
|
const res = await fetch(url, { method: 'HEAD', headers: { 'user-agent': UA } });
|
||||||
|
if (!res.ok) throw new Error(`${res.status} ${url}`);
|
||||||
|
return res.headers.get('last-modified') ?? '';
|
||||||
|
}
|
||||||
|
|
||||||
|
async function importKaikki(lang: string) {
|
||||||
|
const res = await fetch(KAIKKI(lang), { headers: { 'user-agent': UA } });
|
||||||
|
if (!res.ok || !res.body) throw new Error(`${res.status} ${KAIKKI(lang)}`);
|
||||||
|
return replace(lang, 'en', async (add, client) => {
|
||||||
|
const glossRows: string[][] = [];
|
||||||
|
for await (const line of createInterface({ input: Readable.fromWeb(res.body as any), crlfDelay: Infinity })) {
|
||||||
|
if (!line) continue;
|
||||||
|
const d = JSON.parse(line), e = kaikkiEntry(d);
|
||||||
|
if (!e.glosses.length && !e.ipa.length) continue;
|
||||||
|
await add(d.word, e);
|
||||||
|
if (lang === 'ur' && !e.formOf) for (const g of glossKeys(e.glosses.slice(0, 3))) glossRows.push([g, d.word]);
|
||||||
|
}
|
||||||
|
await insertGlosses(client, 'en', glossRows);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
async function insertGlosses(client: any, source: string, rows: string[][]) {
|
||||||
|
for (let i = 0; i < rows.length; i += 1000) {
|
||||||
|
const chunk = rows.slice(i, i + 1000);
|
||||||
|
await client.query(`INSERT INTO ur_glosses (gloss, word, source) VALUES ${chunk.map((_, j) => `($${j * 2 + 1},$${j * 2 + 2},'${source}')`).join(',')}`, chunk.flat());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async function importDump(lang: string) {
|
||||||
|
const bz = spawn('sh', ['-c', `curl -sfL -A '${UA}' '${DUMP(lang)}' | bzcat`]);
|
||||||
|
return replace(lang, 'own', async (add, client) => {
|
||||||
|
let buf = '';
|
||||||
|
const glossRows: string[][] = []; // ur.wiktionary English entries -> the pivot table
|
||||||
|
for await (const chunk of bz.stdout.setEncoding('utf8')) {
|
||||||
|
buf += chunk;
|
||||||
|
const end = buf.lastIndexOf('</page>');
|
||||||
|
if (end < 0) continue;
|
||||||
|
for (const p of dumpPages(buf.slice(0, end + 7))) {
|
||||||
|
if (lang === 'ur' && isEnglish(p.title)) {
|
||||||
|
for (const w of englishToUrdu(p.text)) glossRows.push([p.title.toLowerCase(), w]);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
let { defs, links } = definitions(wt, code), title = f;
|
const e = ownEntry(lang, p.text);
|
||||||
if (!defs.length && links[0]) ({ defs } = definitions(await wikitext(host, (title = links[0])), code));
|
if (e.defs.length || e.links.length || 'english' in e) await add(p.title, e);
|
||||||
if (defs.length) return { defs, origin: null, url: `https://${host}/wiki/${encodeURIComponent(title)}` };
|
|
||||||
}
|
}
|
||||||
return null;
|
buf = buf.slice(end + 7);
|
||||||
}
|
|
||||||
|
|
||||||
// ur.wiktionary: numbered lines under ==معانی==, origin from "(عربی)" on the first line
|
|
||||||
export function urduEntry(wt: string) {
|
|
||||||
const m = wt.split(/==\s*معانی\s*==/)[1]?.split(/\n==/)[0] ?? '';
|
|
||||||
const meanings = m.split('\n').map((l) => l.match(/^\s*(?:\d+[.)-]|#)\s*(.+)/)?.[1]).filter(Boolean).map((l) => unlink(l!)).slice(0, 6);
|
|
||||||
const origin = unlink(wt.match(/\((\[\[[^\]]+\]\])\)/)?.[1] ?? '') || null;
|
|
||||||
return { meanings, origin };
|
|
||||||
}
|
|
||||||
|
|
||||||
// en.wiktionary page HTML: per language, IPA, headword transliteration, audio
|
|
||||||
export function pronunciations(html: string) {
|
|
||||||
const out: Record<string, { ipa: string[]; tr: string | null; audio: string | null }> = {};
|
|
||||||
const parts = html.split(/<h2[^>]*id="([^"]+)"/);
|
|
||||||
for (let i = 1; i < parts.length; i += 2) {
|
|
||||||
const code = Object.entries(LANGS).find(([, n]) => n === parts[i])?.[0];
|
|
||||||
if (!code) continue;
|
|
||||||
const body = parts[i + 1];
|
|
||||||
const ipa = [...new Set([...body.matchAll(/class="IPA[^"]*"[^>]*>([^<]+)/g)].map((m) => m[1]).filter((x) => /^[/[]/.test(x)))].slice(0, 2);
|
|
||||||
const tr = body.match(new RegExp(`lang="${code}-Latn" class="headword-tr[^"]*"[^>]*>([^<]+)`))?.[1] ?? null;
|
|
||||||
const a = body.match(/"(?:https:)?(\/\/upload\.wikimedia\.org\/[^"]+\.(?:ogg|oga|mp3|wav))"/)?.[1];
|
|
||||||
out[code] = { ipa, tr: tr && plain(tr), audio: a ? 'https:' + a : null };
|
|
||||||
}
|
}
|
||||||
return out;
|
const code: number = await new Promise((r) => (bz.exitCode !== null ? r(bz.exitCode) : bz.on('close', r)));
|
||||||
|
if (code !== 0) throw new Error(`dump download/decompress failed for ${lang} (exit ${code})`);
|
||||||
|
await insertGlosses(client, 'own', glossRows);
|
||||||
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
// Urdu equivalents of single-word English glosses, from their translation tables
|
// changes on ur./fa./ar.wiktionary since the last run, re-read from the API (titles in batches of 50)
|
||||||
async function pivot(word: string, glosses: string[]) {
|
async function recentChanges(lang: string) {
|
||||||
const seen = new Set([word.replace(DIAC, '')]); // the word itself, and duplicates differing only in diacritics
|
const api = `https://${lang}.wiktionary.org/w/api.php`;
|
||||||
const words = [...new Set(glosses.flatMap((g) => g.split(/[,;]/)).map((s) => s.trim()).filter((s) => /^[a-z]+$/.test(s)))].slice(0, 3); // lowercase: no proper nouns
|
const since = await meta(`rc:${lang}`), now = new Date().toISOString();
|
||||||
const found = await Promise.all(words.map(async (gloss) => {
|
if (!since) return setMeta(`rc:${lang}`, now).then(() => 0); // first run: the dump is the baseline
|
||||||
const wt = await wikitext('en.wiktionary.org', gloss);
|
const titles = new Set<string>();
|
||||||
const lines = [...wt.matchAll(/^\*:? Urdu: (.*)$/gm)].map((m) => m[1]).join(' ');
|
let cont: Record<string, string> = {};
|
||||||
const urdu = [...lines.matchAll(/\{\{t\+?\|ur\|([^|}]+)/g)].map((m) => m[1].trim())
|
do {
|
||||||
.filter((u) => !seen.has(u.replace(DIAC, '')) && seen.add(u.replace(DIAC, ''))).slice(0, 8);
|
const q = new URLSearchParams({ action: 'query', list: 'recentchanges', rcnamespace: '0', rctype: 'edit|new', rcprop: 'title',
|
||||||
return { gloss, urdu };
|
rclimit: '500', rcdir: 'newer', rcstart: since, rcend: now, format: 'json', formatversion: '2', ...cont });
|
||||||
}));
|
const d: any = await (await fetch(`${api}?${q}`, { headers: { 'user-agent': UA } })).json();
|
||||||
return found.filter((f) => f.urdu.length);
|
for (const c of d.query?.recentchanges ?? []) titles.add(c.title);
|
||||||
|
cont = d.continue ?? {};
|
||||||
|
} while (cont.rccontinue);
|
||||||
|
const list = [...titles];
|
||||||
|
for (let i = 0; i < list.length; i += 50) {
|
||||||
|
const q = new URLSearchParams({ action: 'query', prop: 'revisions', rvprop: 'content', rvslots: 'main',
|
||||||
|
titles: list.slice(i, i + 50).join('|'), format: 'json', formatversion: '2' });
|
||||||
|
const d: any = await (await fetch(api, { method: 'POST', body: q, headers: { 'user-agent': UA } })).json();
|
||||||
|
for (const p of d.query?.pages ?? []) {
|
||||||
|
const wt = p.revisions?.[0]?.slots?.main?.content;
|
||||||
|
if (lang === 'ur' && isEnglish(p.title)) {
|
||||||
|
const g = p.title.toLowerCase();
|
||||||
|
await pool.query(`DELETE FROM ur_glosses WHERE source = 'own' AND gloss = $1`, [g]);
|
||||||
|
for (const w of wt ? englishToUrdu(wt) : []) await pool.query(`INSERT INTO ur_glosses (gloss, word, source) VALUES ($1, $2, 'own')`, [g, w]);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
await pool.query(`DELETE FROM wiktionary WHERE lang = $1 AND source = 'own' AND title = $2`, [lang, p.title]);
|
||||||
|
const e = wt && !/^#(REDIRECT|تحويل|تغییر)/i.test(wt) ? ownEntry(lang, wt) : null;
|
||||||
|
if (e && (e.defs.length || e.links.length || 'english' in e))
|
||||||
|
await pool.query(`INSERT INTO wiktionary (lang, source, title, key, data) VALUES ($1, 'own', $2, $3, $4)`, [lang, p.title, key(p.title), e]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
await setMeta(`rc:${lang}`, now);
|
||||||
|
return list.length;
|
||||||
}
|
}
|
||||||
|
|
||||||
async function fetchWord(word: string) {
|
// full re-import of each source whose upstream file changed, then the daily recent changes
|
||||||
let title: string | null = null, defs: any = null;
|
export async function sync(log = console.log) {
|
||||||
for (const f of forms(word)) if ((defs = await get(rest('definition', f)))) { title = f; break; }
|
for (const lang of LANGS) {
|
||||||
const [pron, ...own] = await Promise.all([
|
for (const [source, url, run] of [['en', KAIKKI(lang), importKaikki], ['own', DUMP(lang), importDump]] as const) {
|
||||||
title ? get(rest('html', title), false).then((h) => pronunciations(h ?? '')) : {},
|
const lm = await lastModified(url), k = `file:${lang}:${source}`;
|
||||||
...Object.keys(LANGS).map((c) => native(c, word)),
|
if (lm && lm === (await meta(k))) continue;
|
||||||
]) as [Record<string, any>, ...(Awaited<ReturnType<typeof native>>)[]];
|
const t = Date.now(), n = await run(lang);
|
||||||
const langs = Object.keys(LANGS).map((code, i) => ({
|
await setMeta(k, lm);
|
||||||
code, ...(pron[code] ?? { ipa: [], tr: null, audio: null }),
|
if (source === 'own') await setMeta(`rc:${lang}`, new Date(lm).toISOString()); // changes after the dump
|
||||||
meanings: own[i]?.defs ?? [], origin: own[i]?.origin ?? null, source: own[i]?.url ?? null,
|
log(`${lang} ${source}: ${n} entries (${Math.round((Date.now() - t) / 1000)} s)`);
|
||||||
senses: (defs?.[code] ?? []).map((e: any) => ({
|
}
|
||||||
pos: e.partOfSpeech, defs: e.definitions.map((d: any) => plain(d.definition)).filter(Boolean).slice(0, 4),
|
log(`${lang} recent changes: ${await recentChanges(lang)} pages`);
|
||||||
})).filter((s: any) => s.defs.length).slice(0, 3),
|
}
|
||||||
})).filter((l) => l.meanings.length || l.senses.length || l.ipa.length || l.tr);
|
|
||||||
const glosses = langs.find((l) => l.senses.length)?.senses[0].defs.slice(0, 2) ?? []; // the primary meaning only
|
|
||||||
const hasUrdu = langs.some((l) => l.code === 'ur' && l.meanings.length);
|
|
||||||
return {
|
|
||||||
word, langs,
|
|
||||||
equivalents: hasUrdu ? [] : await pivot(word, glosses),
|
|
||||||
sources: { en: title && `https://en.wiktionary.org/wiki/${encodeURIComponent(title)}` },
|
|
||||||
found: langs.length > 0,
|
|
||||||
};
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- lookup ----
|
||||||
|
|
||||||
|
const page = (lang: string, source: string, title: string) =>
|
||||||
|
source === 'en' ? `https://en.wiktionary.org/wiki/${encodeURIComponent(title)}#${NAME[lang as 'ur']}` : `https://${lang}.wiktionary.org/wiki/${encodeURIComponent(title)}`;
|
||||||
|
|
||||||
export async function lookup(word: string) {
|
export async function lookup(word: string) {
|
||||||
const hit = await pool.query(`SELECT data FROM dictionary WHERE word = $1 AND fetched_at > now() - interval '30 days'`, [word]);
|
const k = key(word);
|
||||||
if (hit.rows[0]) return hit.rows[0].data;
|
const { rows } = await pool.query('SELECT lang, source, title, data FROM wiktionary WHERE key = $1 ORDER BY id', [k]);
|
||||||
const data = await fetchWord(word);
|
// ar.wiktionary pointer pages: follow to the diacritised entries
|
||||||
if (!data.found) return data; // not found (or Wiktionary unreachable): don't cache
|
const pointers = rows.filter((r) => r.source === 'own' && !r.data.defs.length).flatMap((r) => r.data.links.map((l: string) => [r.lang, l]));
|
||||||
await pool.query(
|
if (pointers.length) {
|
||||||
`INSERT INTO dictionary (word, data) VALUES ($1, $2) ON CONFLICT (word) DO UPDATE SET data = $2, fetched_at = now()`,
|
const more = await pool.query(
|
||||||
[word, data],
|
`SELECT lang, source, title, data FROM wiktionary WHERE source = 'own' AND (lang, title) IN (SELECT * FROM unnest($1::text[], $2::text[]))`,
|
||||||
);
|
[pointers.map((p) => p[0]), pointers.map((p) => p[1])]);
|
||||||
return data;
|
rows.push(...more.rows);
|
||||||
|
}
|
||||||
|
const langs = LANGS.map((code) => {
|
||||||
|
const en = rows.filter((r) => r.lang === code && r.source === 'en');
|
||||||
|
const own = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.defs.length);
|
||||||
|
const lemmas = en.filter((r) => !r.data.formOf), use = lemmas.length ? lemmas : en;
|
||||||
|
const first = (f: string) => use.map((r) => r.data[f]).find((v) => (Array.isArray(v) ? v.length : v)) ?? null;
|
||||||
|
// ur.wiktionary's English translation lines ("انگریزی : …")
|
||||||
|
const translated = rows.filter((r) => r.lang === code && r.source === 'own' && r.data.english).map((r) => r.data.english);
|
||||||
|
return {
|
||||||
|
code, ipa: first('ipa') ?? [], tr: first('tr'), audio: first('audio'), ety: first('ety'),
|
||||||
|
synonyms: [...new Set(use.flatMap((r) => r.data.synonyms))].slice(0, 8),
|
||||||
|
meanings: own.flatMap((r) => r.data.defs).slice(0, 6), origin: own.find((r) => r.data.origin)?.data.origin ?? null,
|
||||||
|
source: own[0] ? page(code, 'own', own[0].title) : null,
|
||||||
|
senses: [...use.map((r) => ({ pos: r.data.pos, defs: r.data.glosses.slice(0, 4) })).filter((s) => s.defs.length).slice(0, 3),
|
||||||
|
...(translated.length ? [{ pos: 'translation', defs: translated }] : [])],
|
||||||
|
en: en[0] ? page(code, 'en', en[0].title) : null,
|
||||||
|
};
|
||||||
|
}).filter((l) => l.meanings.length || l.senses.length || l.ipa.length || l.tr);
|
||||||
|
// no Urdu meaning: Urdu words sharing the primary English meaning
|
||||||
|
let equivalents: { gloss: string; urdu: string[] }[] = [];
|
||||||
|
if (!langs.some((l) => l.code === 'ur' && l.meanings.length)) {
|
||||||
|
const glosses = glossKeys(langs.find((l) => l.senses.length)?.senses[0].defs.slice(0, 2) ?? []).slice(0, 4);
|
||||||
|
if (glosses.length) {
|
||||||
|
const { rows: eq } = await pool.query('SELECT gloss, word FROM ur_glosses WHERE gloss = ANY($1)', [glosses]);
|
||||||
|
const seen = new Set([k]);
|
||||||
|
equivalents = glosses.map((g) => ({
|
||||||
|
gloss: g, urdu: eq.filter((r) => r.gloss === g && !seen.has(key(r.word)) && seen.add(key(r.word))).map((r) => r.word).slice(0, 8),
|
||||||
|
})).filter((e) => e.urdu.length);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return { word, langs, equivalents, found: langs.length > 0 };
|
||||||
}
|
}
|
||||||
|
|||||||
@ -58,9 +58,25 @@ CREATE TABLE IF NOT EXISTS verses (
|
|||||||
PRIMARY KEY (poem_id, vorder)
|
PRIMARY KEY (poem_id, vorder)
|
||||||
);
|
);
|
||||||
|
|
||||||
-- Wiktionary lookups for the reading sidebar (api/src/dictionary.ts), refreshed after 30 days
|
-- Wiktionary for the word sidebar (api/src/dictionary.ts; filled and kept current by dict-sync.ts)
|
||||||
CREATE TABLE IF NOT EXISTS dictionary (
|
DROP TABLE IF EXISTS dictionary; -- the earlier live-lookup cache
|
||||||
word text PRIMARY KEY,
|
CREATE TABLE IF NOT EXISTS wiktionary (
|
||||||
data jsonb NOT NULL,
|
id bigserial PRIMARY KEY,
|
||||||
fetched_at timestamptz NOT NULL DEFAULT now()
|
lang text NOT NULL, -- the word's language: ur | fa | ar
|
||||||
|
source text NOT NULL, -- en (en.wiktionary, via kaikki.org) | own (that language's Wiktionary)
|
||||||
|
title text NOT NULL, -- headword / page title
|
||||||
|
key text NOT NULL, -- spelling-insensitive lookup key (dictionary.ts key())
|
||||||
|
data jsonb NOT NULL
|
||||||
|
);
|
||||||
|
CREATE INDEX IF NOT EXISTS wiktionary_key ON wiktionary(key);
|
||||||
|
CREATE INDEX IF NOT EXISTS wiktionary_title ON wiktionary(lang, source, title);
|
||||||
|
CREATE TABLE IF NOT EXISTS ur_glosses ( -- English gloss -> Urdu word, for the pivot through English
|
||||||
|
gloss text NOT NULL,
|
||||||
|
word text NOT NULL
|
||||||
|
);
|
||||||
|
ALTER TABLE ur_glosses ADD COLUMN IF NOT EXISTS source text NOT NULL DEFAULT 'en'; -- en | own (ur.wiktionary English entries)
|
||||||
|
CREATE INDEX IF NOT EXISTS ur_glosses_gloss ON ur_glosses(gloss);
|
||||||
|
CREATE TABLE IF NOT EXISTS dict_meta ( -- upstream file versions and recent-changes timestamps
|
||||||
|
name text PRIMARY KEY,
|
||||||
|
value text NOT NULL
|
||||||
);
|
);
|
||||||
|
|||||||
@ -4,6 +4,7 @@
|
|||||||
# 1. update the divan-data checkout and fetch new/edited works from Wikisource (incremental)
|
# 1. update the divan-data checkout and fetch new/edited works from Wikisource (incremental)
|
||||||
# 2. rebuild its search index and site export
|
# 2. rebuild its search index and site export
|
||||||
# 3. load the export into PostgreSQL (upserts; the site reads it live)
|
# 3. load the export into PostgreSQL (upserts; the site reads it live)
|
||||||
|
# 3b. sync the word dictionary from Wiktionary (api/src/dict-sync.ts; first run imports ~700 MB)
|
||||||
# 4. optionally commit + push the refreshed data (DIVAN_DATA_PUSH=1, needs git push access)
|
# 4. optionally commit + push the refreshed data (DIVAN_DATA_PUSH=1, needs git push access)
|
||||||
#
|
#
|
||||||
# Schedule with cron, e.g. daily at 03:15:
|
# Schedule with cron, e.g. daily at 03:15:
|
||||||
@ -11,7 +12,7 @@
|
|||||||
#
|
#
|
||||||
# Settings (env): DIVAN_DATA_DIR (default /opt/divan-data), DIVAN_APP_DIR (default: this repo),
|
# Settings (env): DIVAN_DATA_DIR (default /opt/divan-data), DIVAN_APP_DIR (default: this repo),
|
||||||
# DATABASE_URL (Postgres, as for the API), DIVAN_DATA_PUSH=1 to push data changes.
|
# DATABASE_URL (Postgres, as for the API), DIVAN_DATA_PUSH=1 to push data changes.
|
||||||
# Needs: git, python3 (stdlib only), node 24+.
|
# Needs: git, python3 (stdlib only), node 24+, curl and bzcat (dictionary dumps).
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
APP_DIR=${DIVAN_APP_DIR:-$(cd "$(dirname "$0")/.." && pwd)}
|
APP_DIR=${DIVAN_APP_DIR:-$(cd "$(dirname "$0")/.." && pwd)}
|
||||||
@ -36,4 +37,8 @@ fi
|
|||||||
cd "$APP_DIR/api"
|
cd "$APP_DIR/api"
|
||||||
[ -d node_modules ] || npm ci --omit=dev --silent
|
[ -d node_modules ] || npm ci --omit=dev --silent
|
||||||
node src/import.ts "$DATA_DIR"
|
node src/import.ts "$DATA_DIR"
|
||||||
|
|
||||||
|
# word dictionary (Wiktionary): re-imports sources that changed upstream, then the day's edits.
|
||||||
|
# A failure here must not fail the content sync.
|
||||||
|
node src/dict-sync.ts || echo "$(date -u +%FT%TZ) dictionary sync failed"
|
||||||
echo "== $(date -u +%FT%TZ) done"
|
echo "== $(date -u +%FT%TZ) done"
|
||||||
|
|||||||
@ -120,7 +120,8 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
|
|||||||
const dict = document.getElementById('dict'), dictBody = dict.querySelector('.dict-body'), main = document.querySelector('main');
|
const dict = document.getElementById('dict'), dictBody = dict.querySelector('.dict-body'), main = document.querySelector('main');
|
||||||
const LANG = { ur: 'اردو', fa: 'فارسی', ar: 'عربی' };
|
const LANG = { ur: 'اردو', fa: 'فارسی', ar: 'عربی' };
|
||||||
const POS = { Noun: 'اسم', 'Proper noun': 'اسم معرفہ', Verb: 'فعل', Adjective: 'صفت', Adverb: 'متعلق فعل', Pronoun: 'ضمیر',
|
const POS = { Noun: 'اسم', 'Proper noun': 'اسم معرفہ', Verb: 'فعل', Adjective: 'صفت', Adverb: 'متعلق فعل', Pronoun: 'ضمیر',
|
||||||
Postposition: 'حرف جار', Preposition: 'حرف جار', Conjunction: 'حرف عطف', Interjection: 'حرف ندا', Particle: 'حرف' };
|
Postposition: 'حرف جار', Preposition: 'حرف جار', Conjunction: 'حرف عطف', Interjection: 'حرف ندا', Particle: 'حرف',
|
||||||
|
Translation: 'ترجمہ (اردو ویکی لغت)', Phrase: 'فقرہ', Proverb: 'کہاوت', Name: 'اسم معرفہ' };
|
||||||
const el = (tag, cls, text, attrs) => {
|
const el = (tag, cls, text, attrs) => {
|
||||||
const e = document.createElement(tag);
|
const e = document.createElement(tag);
|
||||||
if (cls) e.className = cls;
|
if (cls) e.className = cls;
|
||||||
@ -134,7 +135,8 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
|
|||||||
current = w; dict.hidden = false; document.body.classList.add('dict-open');
|
current = w; dict.hidden = false; document.body.classList.add('dict-open');
|
||||||
dictBody.replaceChildren(el('h2', null, w), el('p', 'muted', 'تلاش جاری ہے…'));
|
dictBody.replaceChildren(el('h2', null, w), el('p', 'muted', 'تلاش جاری ہے…'));
|
||||||
let d = null;
|
let d = null;
|
||||||
try { const r = await fetch('/api/word?w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {}
|
// v: bump when the response format changes (responses are browser-cached for a day)
|
||||||
|
try { const r = await fetch('/api/word?v=2&w=' + encodeURIComponent(w)); if (r.ok) d = await r.json(); } catch (e) {}
|
||||||
if (w !== current) return; // a newer selection won
|
if (w !== current) return; // a newer selection won
|
||||||
const out = [el('h2', null, w)];
|
const out = [el('h2', null, w)];
|
||||||
if (!d || !d.found) {
|
if (!d || !d.found) {
|
||||||
@ -171,14 +173,20 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک
|
|||||||
if (l.source) label.append(el('a', null, 'ماخذ', { href: l.source, target: '_blank', rel: 'noopener' }));
|
if (l.source) label.append(el('a', null, 'ماخذ', { href: l.source, target: '_blank', rel: 'noopener' }));
|
||||||
sec.append(label, list(l.meanings, null, { lang: l.code }));
|
sec.append(label, list(l.meanings, null, { lang: l.code }));
|
||||||
}
|
}
|
||||||
for (const s of l.senses) sec.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' }));
|
for (const s of l.senses) sec.append(el('p', 'pos', `انگریزی میں · ${POS[s.pos.charAt(0).toUpperCase() + s.pos.slice(1)] ?? s.pos}`), list(s.defs, 'en', { dir: 'ltr', lang: 'en' }));
|
||||||
|
if (l.synonyms.length) sec.append(el('p', 'pos', 'مترادفات'), el('p', null, l.synonyms.join('، ')));
|
||||||
|
if (l.ety) sec.append(el('p', 'pos', 'اشتقاق'), el('p', 'en ety', l.ety, { dir: 'ltr', lang: 'en' }));
|
||||||
|
if (l.en) {
|
||||||
|
const p = el('p', 'pos');
|
||||||
|
p.append(el('a', null, 'انگریزی ویکی لغت', { href: l.en, target: '_blank', rel: 'noopener' }));
|
||||||
|
sec.append(p);
|
||||||
|
}
|
||||||
out.push(sec);
|
out.push(sec);
|
||||||
if (l.code === 'ur' && d.equivalents.length) out.push(equivalents());
|
if (l.code === 'ur' && d.equivalents.length) out.push(equivalents());
|
||||||
}
|
}
|
||||||
if (d.equivalents.length && !d.langs.some((l) => l.code === 'ur')) out.splice(1, 0, equivalents());
|
if (d.equivalents.length && !d.langs.some((l) => l.code === 'ur')) out.splice(1, 0, equivalents());
|
||||||
const src = el('p', 'src muted');
|
const src = el('p', 'src muted');
|
||||||
src.append('ویکی لغت · ');
|
src.append('ویکی لغت (Wiktionary) · ');
|
||||||
if (d.sources.en) src.append(el('a', null, 'انگریزی ویکی لغت', { href: d.sources.en, target: '_blank', rel: 'noopener' }), ' · ');
|
|
||||||
src.append('CC BY-SA');
|
src.append('CC BY-SA');
|
||||||
dictBody.replaceChildren(...out, src);
|
dictBody.replaceChildren(...out, src);
|
||||||
}
|
}
|
||||||
|
|||||||
@ -146,6 +146,7 @@ h1 + .muted { text-align: center; margin-top: 0; }
|
|||||||
.dict .pos { font-size: .8rem; color: var(--muted); margin: 8px 0 0; }
|
.dict .pos { font-size: .8rem; color: var(--muted); margin: 8px 0 0; }
|
||||||
.dict .note { font-size: .8rem; margin: 0; }
|
.dict .note { font-size: .8rem; margin: 0; }
|
||||||
.dict .src { font-size: .78rem; margin-top: 18px; }
|
.dict .src { font-size: .78rem; margin-top: 18px; }
|
||||||
|
.dict .ety { text-align: left; color: var(--muted); font-size: .8rem; }
|
||||||
.dict-close { position: sticky; top: 0; float: left; border: 0; background: none; color: var(--muted); font-size: 1rem; cursor: pointer; }
|
.dict-close { position: sticky; top: 0; float: left; border: 0; background: none; color: var(--muted); font-size: 1rem; cursor: pointer; }
|
||||||
.dict .play { border: 1px solid var(--border); background: var(--inner); border-radius: 8px; cursor: pointer; padding: 0 6px; }
|
.dict .play { border: 1px solid var(--border); background: var(--inner); border-radius: 8px; cursor: pointer; padding: 0 6px; }
|
||||||
/* wide screens: the page moves over instead of being covered */
|
/* wide screens: the page moves over instead of being covered */
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user