Search: highlighted matches, matching couplet or prose excerpt, exact phrase first; opened pages keep the highlight (?hl=)

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Anas Rashid 2026-10-08 21:30:42 +02:00
parent 9a2db49eab
commit efb7aacde3
9 changed files with 109 additions and 22 deletions

View File

@ -50,7 +50,7 @@ Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DI
|---|---|
| `GET /api/poets` | all poets |
| `GET /api/page?url=/p238/...` | the poet, category or poem at a site URL (breadcrumbs, children, verses, prev/next) |
| `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised |
| `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised; exact phrase first; each with the best-matching couplet or paragraph (`snippet`) |
| `GET /health` | database check |
Search normalises both stored text and queries: Arabic ي/ك/ه → Urdu ی/ک/ہ, ۂ/ۓ, diacritics and the Urdu full stop removed; do-chashmi ھ stays distinct.

View File

@ -5,7 +5,7 @@
// GET /health
import Fastify from 'fastify';
import { pool } from './db.ts';
import { likePatterns } from './urdu.ts';
import { likePatterns, normalise, terms } from './urdu.ts';
const app = Fastify({ logger: { level: process.env.LOG_LEVEL ?? 'info' } });
const PAGE_SIZE = 20;
@ -91,16 +91,37 @@ app.get<{ Querystring: { q?: string; poet?: string; page?: string } }>('/api/sea
where.push(`p.poet_id = $${params.length}`);
}
const sql = `FROM poems p JOIN poets t ON t.id = p.poet_id WHERE ${where.join(' AND ')}`;
// several words: poems with them together as a phrase come first
const phrase = likePatterns(`"${req.query.q}"`)[0];
const [count, rows] = await Promise.all([
pool.query(`SELECT count(*)::int AS n ${sql}`, params),
pool.query(
`SELECT p.url, p.title, t.nickname AS poet, t.url AS poet_url,
(SELECT v.text FROM verses v WHERE v.poem_id = p.id ORDER BY v.vorder LIMIT 1) AS first_line
${sql} ORDER BY t.birth_year_ah NULLS LAST, p.id LIMIT ${PAGE_SIZE} OFFSET ${(page - 1) * PAGE_SIZE}`,
params,
`SELECT p.id, p.url, p.title, t.nickname AS poet, t.url AS poet_url
${sql} ORDER BY p.search_text ILIKE $${params.length + 1} DESC, t.birth_year_ah NULLS LAST, p.id
LIMIT ${PAGE_SIZE} OFFSET ${(page - 1) * PAGE_SIZE}`,
[...params, phrase],
),
]);
return { total: count.rows[0].n, page, pageSize: PAGE_SIZE, results: rows.rows };
// snippet: the verse holding most of the terms, best the whole phrase (with its couplet partner), else the first line
const ts = terms(req.query.q ?? ''), whole = ts.join(' '), top = ts.length + (ts.length > 1 ? 1 : 0);
const verses = await pool.query(
'SELECT poem_id, position, couplet, text FROM verses WHERE poem_id = ANY($1) ORDER BY poem_id, vorder',
[rows.rows.map((r) => r.id)],
);
const byPoem = Map.groupBy(verses.rows, (v) => v.poem_id);
const results = rows.rows.map(({ id, ...r }) => {
const vs = byPoem.get(id) ?? [];
let best = vs[0], score = 0;
for (const v of vs) {
const n = normalise(v.text), k = ts.filter((t) => n.includes(t)).length + (ts.length > 1 && n.includes(whole) ? 1 : 0);
if (k > score) [best, score] = [v, k];
if (score === top) break;
}
const lines = !best ? [] : best.position === 'Paragraph' || best.position === 'Single' ? [best.text]
: vs.filter((v) => v.couplet === best.couplet && (v.position === 'Right' || v.position === 'Left')).map((v) => v.text);
return { ...r, snippet: lines, prose: best?.position === 'Paragraph' };
});
return { total: count.rows[0].n, page, pageSize: PAGE_SIZE, results };
});
const port = Number(process.env.PORT ?? 4100);

View File

@ -1,6 +1,6 @@
import { test } from 'node:test';
import assert from 'node:assert/strict';
import { normalise, likePatterns } from './urdu.ts';
import { normalise, likePatterns, highlighter } from './urdu.ts';
test('Urdu variants and diacritics normalise to the same form', () => {
const same = (a: string, b: string) => assert.equal(normalise(a), normalise(b), `${a} vs ${b}`);
@ -18,3 +18,13 @@ test('search patterns: words, phrases, escaping, empty', () => {
assert.deepEqual(likePatterns('50%_x'), ['%50\\%\\_x%']);
assert.deepEqual(likePatterns(' '), []);
});
test('highlighter finds terms in original text despite variants and diacritics', () => {
const marks = (q: string, text: string) => text.split(highlighter(q)!).filter((_, i) => i % 2);
assert.deepEqual(marks('دل ناداں', 'دلِ ناداں تجھے ہوا کیا ہے'), ['دلِ', 'ناداں']);
assert.deepEqual(marks('"دل ناداں"', 'دلِ ناداں تجھے'), ['دلِ ناداں']);
assert.deepEqual(marks('کوئی', 'كوئي اميد بر نہيں آتی'), ['كوئي']);
assert.deepEqual(marks('نالہ', 'نالۂ دل'), ['نالۂ']);
assert.deepEqual(marks('کہ', 'کچھ کھ کہ'), ['کہ']); // do-chashmi stays distinct
assert.equal(highlighter(' '), null);
});

View File

@ -33,12 +33,29 @@ export function normalise(text: string): string {
.trim();
}
// ILIKE patterns for a search term: "quoted phrase" -> one pattern, otherwise one per word (all must match).
export function likePatterns(term: string): string[] {
// Normalised search terms: "quoted phrase" -> one term, otherwise one per word (all must match).
export function terms(term: string): string[] {
const t = (term ?? '').trim();
if (!t) return [];
const phrase = t.length > 1 && t.startsWith('"') && t.endsWith('"');
const n = normalise(t);
const parts = phrase ? [n] : n.split(' ');
return parts.filter(Boolean).map((p) => '%' + p.replace(/[\\%_]/g, (c) => '\\' + c) + '%');
return (phrase ? [n] : n.split(' ')).filter(Boolean);
}
// ILIKE patterns, one per term.
export const likePatterns = (term: string) => terms(term).map((p) => '%' + p.replace(/[\\%_]/g, (c) => '\\' + c) + '%');
// A regex that finds the terms in original (unnormalised) text, with one capture group so
// text.split(re) puts the matches at odd indexes. Each letter also matches its variants, with
// diacritics allowed between letters; a space matches spaces and punctuation.
const VARIANTS: Record<string, string> = {};
for (const [from, to] of Object.entries(LETTERS)) VARIANTS[to] = (VARIANTS[to] ?? to) + from;
const SKIP = '[\u064B-\u0652\u0654\u0670\u0657\u0658\u0615\u200D\u200E\u200F]*';
export function highlighter(term: string): RegExp | null {
const ts = terms(term);
if (!ts.length) return null;
const one = (t: string) => [...t].map((c) =>
c === ' ' ? '[\\s\u200C.\u060C!\u061F:\u061B;*()"\'\u00AB\u00BB\u06D4]+'
: VARIANTS[c] ? `[${VARIANTS[c]}]` : c.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')).join(SKIP) + SKIP;
return new RegExp(`(${ts.sort((a, b) => b.length - a.length).map(one).join('|')})`, 'iu');
}

View File

@ -0,0 +1,5 @@
---
// renders text segments: [text, 'mark' (search hit) | 'pen' (takhallus) | '']
const { s } = Astro.props as { s: [string, string][] };
---
{s.map(([t, k]) => (k === 'mark' ? <mark>{t}</mark> : k === 'pen' ? <span class="takhallus">{t}</span> : t))}

View File

@ -13,3 +13,15 @@ const span = (a: number | null, b: number | null, mark: string) =>
// Hijri and Gregorian (عیسوی): "۱۲۹۴ – ۱۳۵۷ ھ · ۱۸۷۷ – ۱۹۳۸ ء"
export const years = (p: { birth_year_ah: number | null; death_year_ah: number | null; birth_year_ce?: number | null; death_year_ce?: number | null }) =>
[span(p.birth_year_ah, p.death_year_ah, 'ھ'), span(p.birth_year_ce ?? null, p.death_year_ce ?? null, 'ء')].filter(Boolean).join(' · ');
// search highlighting (same matching rules as the API's search)
export { highlighter } from '../../../api/src/urdu.ts';
// text split so odd indexes are the matches
export const parts = (text: string, re: RegExp | null) => (re ? text.split(re) : [text]);
// prose: a stretch of text around the first match
export function excerpt(text: string, re: RegExp | null, room = 90) {
if (text.length <= room * 2) return text;
const i = Math.max(0, re ? text.search(re) : 0);
const a = Math.max(0, text.lastIndexOf(' ', Math.max(0, i - room))), b = text.indexOf(' ', i + room);
return (a > 0 ? '… ' : '') + text.slice(a, b < 0 ? undefined : b).trim() + (b < 0 ? '' : ' …');
}

View File

@ -1,8 +1,9 @@
---
// Poet (/p238), category (/p238/shaeri/c641) and poem (/p238/shaeri/c641/sh6556) pages
import Base from '../layouts/Base.astro';
import Seg from '../components/Seg.astro';
import { api } from '../lib/api';
import { ud, years } from '../lib/urdu';
import { ud, years, highlighter, parts } from '../lib/urdu';
const url = '/' + (Astro.params.path ?? '');
const page = await api(`/api/page?url=${encodeURIComponent(url)}`);
@ -22,7 +23,11 @@ if (page.type === 'poem') {
}
}
// pen name (marked ؔ in the text) set apart; split() with a capture group puts matches at odd indexes
const verse = (line: string) => line.split(/(\S*\u0614\S*)/);
// ?hl= (from search results) marks the searched words on the page; segments: [text, 'mark' | 'pen' | '']
const hlq = Astro.url.searchParams.get('hl') ?? '';
const re = highlighter(hlq);
const show = (line: string) => parts(line, re).flatMap((s, i): [string, string][] =>
i % 2 ? [[s, 'mark']] : s.split(/(\S*\u0614\S*)/).map((t, j) => [t, j % 2 ? 'pen' : '']));
// contents of a divan grouped by radif letter ("ردیف الف … ی"); ے is filed under ی as in printed divans
const LETTER = (l: string) => (l === 'ا' ? 'الف' : l === 'ے' ? 'ی' : l);
const groups: { letter: string | null; poems: any[] }[] = [];
@ -68,13 +73,18 @@ const title = page.type === 'poem' ? page.poem.title : page.type === 'poet' ? pa
{page.type === 'poem' && (
<>
<h1>{page.poem.title}</h1>
<h1><Seg s={show(page.poem.title)} /></h1>
{re && (
<p class="hl-note muted">
«{hlq}» نمایاں · <a href={`/search?q=${encodeURIComponent(hlq)}`}>تلاش کے نتائج</a> · <a href={url}>نشان ہٹائیں</a>
</p>
)}
{page.poem.radif != null && <p class="muted radif">{page.poem.radif ? `ردیف: ${page.poem.radif}` : 'غیر مردف'}</p>}
<article class="poem">
{blocks.map((b) =>
b.kind === 'couplet' ? <div class="couplet" data-mark={b.mark}>{b.lines.map((l) => <p>{verse(l).map((s, i) => i % 2 ? <span class="takhallus">{s}</span> : s)}</p>)}</div>
: b.kind === 'para' ? <p class="para">{b.lines[0]}</p>
: <p class="single">{b.lines[0]}</p>
b.kind === 'couplet' ? <div class="couplet" data-mark={b.mark}>{b.lines.map((l) => <p><Seg s={show(l)} /></p>)}</div>
: b.kind === 'para' ? <p class="para"><Seg s={show(b.lines[0])} /></p>
: <p class="single"><Seg s={show(b.lines[0])} /></p>
)}
</article>
<nav class="poem-nav">
@ -86,4 +96,5 @@ const title = page.type === 'poem' ? page.poem.title : page.type === 'poet' ? pa
)}
</>
)}
{re && <script is:inline>document.querySelector('.poem mark')?.scrollIntoView({ block: 'center' });</script>}
</Base>

View File

@ -1,12 +1,16 @@
---
import Base from '../layouts/Base.astro';
import Seg from '../components/Seg.astro';
import { api } from '../lib/api';
import { ud } from '../lib/urdu';
import { ud, highlighter, parts, excerpt } from '../lib/urdu';
const q = Astro.url.searchParams.get('q')?.trim() ?? '';
const pageNo = Math.max(1, Number(Astro.url.searchParams.get('page')) || 1);
const res = q ? await api(`/api/search?q=${encodeURIComponent(q)}&page=${pageNo}`) : null;
const pages = res ? Math.ceil(res.total / res.pageSize) : 0;
const re = highlighter(q);
const hl = (t: string) => parts(t, re).map((s, i): [string, string] => [s, i % 2 ? 'mark' : '']);
const open = (url: string) => `${url}?hl=${encodeURIComponent(q)}`;
const link = (n: number) => `/search?q=${encodeURIComponent(q)}&page=${n}`;
---
<Base title={q ? `تلاش: ${q}` : 'تلاش'} q={q}>
@ -17,8 +21,12 @@ const link = (n: number) => `/search?q=${encodeURIComponent(q)}&page=${n}`;
<p class="muted">{res.total ? `${ud(res.total)} نتائج` : 'کوئی نتیجہ نہیں ملا۔ کم الفاظ یا کلیدی لفظ آزمائیں۔'}</p>
{res.results.map((r: any) => (
<div class="result">
<a href={r.url}>{r.title}</a> · <a class="muted" href={r.poet_url}>{r.poet}</a>
{r.first_line && <div class="first-line">{r.first_line}</div>}
<a href={open(r.url)}><Seg s={hl(r.title)} /></a> · <a class="muted" href={r.poet_url}>{r.poet}</a>
{r.snippet.length > 0 && (
<a class="snippet" href={open(r.url)}>
{r.prose ? <p><Seg s={hl(excerpt(r.snippet[0], re))} /></p> : r.snippet.map((l: string) => <p><Seg s={hl(l)} /></p>)}
</a>
)}
</div>
))}
{pages > 1 && (

View File

@ -105,7 +105,10 @@ h1::after { content: ''; display: block; width: 140px; height: 2px; background:
/* search */
.result { border-bottom: 1px dashed var(--gold-light); padding: 8px 0; }
.result .first-line { color: var(--muted); }
.result .snippet { display: block; color: var(--ink); margin-top: 4px; padding: 4px 12px; border-inline-start: 2px solid var(--gold-light); }
.result .snippet p { margin: 0; }
mark { background: var(--gold-subtle); color: inherit; border-radius: 3px; padding: 0 2px; box-shadow: inset 0 -2px var(--gold); }
.hl-note { text-align: center; font-size: .85rem; }
.pager { display: flex; gap: 12px; justify-content: center; margin-top: 20px; }
.site-footer { border-top: 1.5px solid var(--border); padding: 16px 0; font-size: .85rem; color: var(--muted); text-align: center; }