diff --git a/README.md b/README.md index 61138322..933dc6f7 100644 --- a/README.md +++ b/README.md @@ -50,7 +50,7 @@ Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DI |---|---| | `GET /api/poets` | all poets | | `GET /api/page?url=/p238/...` | the poet, category or poem at a site URL (breadcrumbs, children, verses, prev/next) | -| `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised | +| `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised; exact phrase first; each with the best-matching couplet or paragraph (`snippet`) | | `GET /health` | database check | Search normalises both stored text and queries: Arabic ي/ك/ه → Urdu ی/ک/ہ, ۂ/ۓ, diacritics and the Urdu full stop removed; do-chashmi ھ stays distinct. diff --git a/api/src/server.ts b/api/src/server.ts index 33899e30..f8e0ce77 100644 --- a/api/src/server.ts +++ b/api/src/server.ts @@ -5,7 +5,7 @@ // GET /health import Fastify from 'fastify'; import { pool } from './db.ts'; -import { likePatterns } from './urdu.ts'; +import { likePatterns, normalise, terms } from './urdu.ts'; const app = Fastify({ logger: { level: process.env.LOG_LEVEL ?? 'info' } }); const PAGE_SIZE = 20; @@ -91,16 +91,37 @@ app.get<{ Querystring: { q?: string; poet?: string; page?: string } }>('/api/sea where.push(`p.poet_id = $${params.length}`); } const sql = `FROM poems p JOIN poets t ON t.id = p.poet_id WHERE ${where.join(' AND ')}`; + // several words: poems with them together as a phrase come first + const phrase = likePatterns(`"${req.query.q}"`)[0]; const [count, rows] = await Promise.all([ pool.query(`SELECT count(*)::int AS n ${sql}`, params), pool.query( - `SELECT p.url, p.title, t.nickname AS poet, t.url AS poet_url, - (SELECT v.text FROM verses v WHERE v.poem_id = p.id ORDER BY v.vorder LIMIT 1) AS first_line - ${sql} ORDER BY t.birth_year_ah NULLS LAST, p.id LIMIT ${PAGE_SIZE} OFFSET ${(page - 1) * PAGE_SIZE}`, - params, + `SELECT p.id, p.url, p.title, t.nickname AS poet, t.url AS poet_url + ${sql} ORDER BY p.search_text ILIKE $${params.length + 1} DESC, t.birth_year_ah NULLS LAST, p.id + LIMIT ${PAGE_SIZE} OFFSET ${(page - 1) * PAGE_SIZE}`, + [...params, phrase], ), ]); - return { total: count.rows[0].n, page, pageSize: PAGE_SIZE, results: rows.rows }; + // snippet: the verse holding most of the terms, best the whole phrase (with its couplet partner), else the first line + const ts = terms(req.query.q ?? ''), whole = ts.join(' '), top = ts.length + (ts.length > 1 ? 1 : 0); + const verses = await pool.query( + 'SELECT poem_id, position, couplet, text FROM verses WHERE poem_id = ANY($1) ORDER BY poem_id, vorder', + [rows.rows.map((r) => r.id)], + ); + const byPoem = Map.groupBy(verses.rows, (v) => v.poem_id); + const results = rows.rows.map(({ id, ...r }) => { + const vs = byPoem.get(id) ?? []; + let best = vs[0], score = 0; + for (const v of vs) { + const n = normalise(v.text), k = ts.filter((t) => n.includes(t)).length + (ts.length > 1 && n.includes(whole) ? 1 : 0); + if (k > score) [best, score] = [v, k]; + if (score === top) break; + } + const lines = !best ? [] : best.position === 'Paragraph' || best.position === 'Single' ? [best.text] + : vs.filter((v) => v.couplet === best.couplet && (v.position === 'Right' || v.position === 'Left')).map((v) => v.text); + return { ...r, snippet: lines, prose: best?.position === 'Paragraph' }; + }); + return { total: count.rows[0].n, page, pageSize: PAGE_SIZE, results }; }); const port = Number(process.env.PORT ?? 4100); diff --git a/api/src/urdu.test.ts b/api/src/urdu.test.ts index 2024e365..3d616b04 100644 --- a/api/src/urdu.test.ts +++ b/api/src/urdu.test.ts @@ -1,6 +1,6 @@ import { test } from 'node:test'; import assert from 'node:assert/strict'; -import { normalise, likePatterns } from './urdu.ts'; +import { normalise, likePatterns, highlighter } from './urdu.ts'; test('Urdu variants and diacritics normalise to the same form', () => { const same = (a: string, b: string) => assert.equal(normalise(a), normalise(b), `${a} vs ${b}`); @@ -18,3 +18,13 @@ test('search patterns: words, phrases, escaping, empty', () => { assert.deepEqual(likePatterns('50%_x'), ['%50\\%\\_x%']); assert.deepEqual(likePatterns(' '), []); }); + +test('highlighter finds terms in original text despite variants and diacritics', () => { + const marks = (q: string, text: string) => text.split(highlighter(q)!).filter((_, i) => i % 2); + assert.deepEqual(marks('دل ناداں', 'دلِ ناداں تجھے ہوا کیا ہے'), ['دلِ', 'ناداں']); + assert.deepEqual(marks('"دل ناداں"', 'دلِ ناداں تجھے'), ['دلِ ناداں']); + assert.deepEqual(marks('کوئی', 'كوئي اميد بر نہيں آتی'), ['كوئي']); + assert.deepEqual(marks('نالہ', 'نالۂ دل'), ['نالۂ']); + assert.deepEqual(marks('کہ', 'کچھ کھ کہ'), ['کہ']); // do-chashmi stays distinct + assert.equal(highlighter(' '), null); +}); diff --git a/api/src/urdu.ts b/api/src/urdu.ts index b07fb0a6..60334f4f 100644 --- a/api/src/urdu.ts +++ b/api/src/urdu.ts @@ -33,12 +33,29 @@ export function normalise(text: string): string { .trim(); } -// ILIKE patterns for a search term: "quoted phrase" -> one pattern, otherwise one per word (all must match). -export function likePatterns(term: string): string[] { +// Normalised search terms: "quoted phrase" -> one term, otherwise one per word (all must match). +export function terms(term: string): string[] { const t = (term ?? '').trim(); if (!t) return []; const phrase = t.length > 1 && t.startsWith('"') && t.endsWith('"'); const n = normalise(t); - const parts = phrase ? [n] : n.split(' '); - return parts.filter(Boolean).map((p) => '%' + p.replace(/[\\%_]/g, (c) => '\\' + c) + '%'); + return (phrase ? [n] : n.split(' ')).filter(Boolean); +} + +// ILIKE patterns, one per term. +export const likePatterns = (term: string) => terms(term).map((p) => '%' + p.replace(/[\\%_]/g, (c) => '\\' + c) + '%'); + +// A regex that finds the terms in original (unnormalised) text, with one capture group so +// text.split(re) puts the matches at odd indexes. Each letter also matches its variants, with +// diacritics allowed between letters; a space matches spaces and punctuation. +const VARIANTS: Record = {}; +for (const [from, to] of Object.entries(LETTERS)) VARIANTS[to] = (VARIANTS[to] ?? to) + from; +const SKIP = '[\u064B-\u0652\u0654\u0670\u0657\u0658\u0615\u200D\u200E\u200F]*'; +export function highlighter(term: string): RegExp | null { + const ts = terms(term); + if (!ts.length) return null; + const one = (t: string) => [...t].map((c) => + c === ' ' ? '[\\s\u200C.\u060C!\u061F:\u061B;*()"\'\u00AB\u00BB\u06D4]+' + : VARIANTS[c] ? `[${VARIANTS[c]}]` : c.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')).join(SKIP) + SKIP; + return new RegExp(`(${ts.sort((a, b) => b.length - a.length).map(one).join('|')})`, 'iu'); } diff --git a/web/src/components/Seg.astro b/web/src/components/Seg.astro new file mode 100644 index 00000000..42fffac3 --- /dev/null +++ b/web/src/components/Seg.astro @@ -0,0 +1,5 @@ +--- +// renders text segments: [text, 'mark' (search hit) | 'pen' (takhallus) | ''] +const { s } = Astro.props as { s: [string, string][] }; +--- +{s.map(([t, k]) => (k === 'mark' ? {t} : k === 'pen' ? {t} : t))} diff --git a/web/src/lib/urdu.ts b/web/src/lib/urdu.ts index 18e3ee3b..50a00dcc 100644 --- a/web/src/lib/urdu.ts +++ b/web/src/lib/urdu.ts @@ -13,3 +13,15 @@ const span = (a: number | null, b: number | null, mark: string) => // Hijri and Gregorian (عیسوی): "۱۲۹۴ – ۱۳۵۷ ھ · ۱۸۷۷ – ۱۹۳۸ ء" export const years = (p: { birth_year_ah: number | null; death_year_ah: number | null; birth_year_ce?: number | null; death_year_ce?: number | null }) => [span(p.birth_year_ah, p.death_year_ah, 'ھ'), span(p.birth_year_ce ?? null, p.death_year_ce ?? null, 'ء')].filter(Boolean).join(' · '); + +// search highlighting (same matching rules as the API's search) +export { highlighter } from '../../../api/src/urdu.ts'; +// text split so odd indexes are the matches +export const parts = (text: string, re: RegExp | null) => (re ? text.split(re) : [text]); +// prose: a stretch of text around the first match +export function excerpt(text: string, re: RegExp | null, room = 90) { + if (text.length <= room * 2) return text; + const i = Math.max(0, re ? text.search(re) : 0); + const a = Math.max(0, text.lastIndexOf(' ', Math.max(0, i - room))), b = text.indexOf(' ', i + room); + return (a > 0 ? '… ' : '') + text.slice(a, b < 0 ? undefined : b).trim() + (b < 0 ? '' : ' …'); +} diff --git a/web/src/pages/[...path].astro b/web/src/pages/[...path].astro index 4163a9ba..12a6d995 100644 --- a/web/src/pages/[...path].astro +++ b/web/src/pages/[...path].astro @@ -1,8 +1,9 @@ --- // Poet (/p238), category (/p238/shaeri/c641) and poem (/p238/shaeri/c641/sh6556) pages import Base from '../layouts/Base.astro'; +import Seg from '../components/Seg.astro'; import { api } from '../lib/api'; -import { ud, years } from '../lib/urdu'; +import { ud, years, highlighter, parts } from '../lib/urdu'; const url = '/' + (Astro.params.path ?? ''); const page = await api(`/api/page?url=${encodeURIComponent(url)}`); @@ -22,7 +23,11 @@ if (page.type === 'poem') { } } // pen name (marked ؔ in the text) set apart; split() with a capture group puts matches at odd indexes -const verse = (line: string) => line.split(/(\S*\u0614\S*)/); +// ?hl= (from search results) marks the searched words on the page; segments: [text, 'mark' | 'pen' | ''] +const hlq = Astro.url.searchParams.get('hl') ?? ''; +const re = highlighter(hlq); +const show = (line: string) => parts(line, re).flatMap((s, i): [string, string][] => + i % 2 ? [[s, 'mark']] : s.split(/(\S*\u0614\S*)/).map((t, j) => [t, j % 2 ? 'pen' : ''])); // contents of a divan grouped by radif letter ("ردیف الف … ی"); ے is filed under ی as in printed divans const LETTER = (l: string) => (l === 'ا' ? 'الف' : l === 'ے' ? 'ی' : l); const groups: { letter: string | null; poems: any[] }[] = []; @@ -68,13 +73,18 @@ const title = page.type === 'poem' ? page.poem.title : page.type === 'poet' ? pa {page.type === 'poem' && ( <> -

{page.poem.title}

+

+ {re && ( +

+ «{hlq}» نمایاں · تلاش کے نتائج · نشان ہٹائیں +

+ )} {page.poem.radif != null &&

{page.poem.radif ? `ردیف: ${page.poem.radif}` : 'غیر مردف'}

}
{blocks.map((b) => - b.kind === 'couplet' ?
{b.lines.map((l) =>

{verse(l).map((s, i) => i % 2 ? {s} : s)}

)}
- : b.kind === 'para' ?

{b.lines[0]}

- :

{b.lines[0]}

+ b.kind === 'couplet' ?
{b.lines.map((l) =>

)}
+ : b.kind === 'para' ?

+ :

)}