Search: highlighted matches, matching couplet or prose excerpt, exact phrase first; opened pages keep the highlight (?hl=)
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
9a2db49eab
commit
efb7aacde3
@ -50,7 +50,7 @@ Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DI
|
||||
|---|---|
|
||||
| `GET /api/poets` | all poets |
|
||||
| `GET /api/page?url=/p238/...` | the poet, category or poem at a site URL (breadcrumbs, children, verses, prev/next) |
|
||||
| `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised |
|
||||
| `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised; exact phrase first; each with the best-matching couplet or paragraph (`snippet`) |
|
||||
| `GET /health` | database check |
|
||||
|
||||
Search normalises both stored text and queries: Arabic ي/ك/ه → Urdu ی/ک/ہ, ۂ/ۓ, diacritics and the Urdu full stop removed; do-chashmi ھ stays distinct.
|
||||
|
||||
@ -5,7 +5,7 @@
|
||||
// GET /health
|
||||
import Fastify from 'fastify';
|
||||
import { pool } from './db.ts';
|
||||
import { likePatterns } from './urdu.ts';
|
||||
import { likePatterns, normalise, terms } from './urdu.ts';
|
||||
|
||||
const app = Fastify({ logger: { level: process.env.LOG_LEVEL ?? 'info' } });
|
||||
const PAGE_SIZE = 20;
|
||||
@ -91,16 +91,37 @@ app.get<{ Querystring: { q?: string; poet?: string; page?: string } }>('/api/sea
|
||||
where.push(`p.poet_id = $${params.length}`);
|
||||
}
|
||||
const sql = `FROM poems p JOIN poets t ON t.id = p.poet_id WHERE ${where.join(' AND ')}`;
|
||||
// several words: poems with them together as a phrase come first
|
||||
const phrase = likePatterns(`"${req.query.q}"`)[0];
|
||||
const [count, rows] = await Promise.all([
|
||||
pool.query(`SELECT count(*)::int AS n ${sql}`, params),
|
||||
pool.query(
|
||||
`SELECT p.url, p.title, t.nickname AS poet, t.url AS poet_url,
|
||||
(SELECT v.text FROM verses v WHERE v.poem_id = p.id ORDER BY v.vorder LIMIT 1) AS first_line
|
||||
${sql} ORDER BY t.birth_year_ah NULLS LAST, p.id LIMIT ${PAGE_SIZE} OFFSET ${(page - 1) * PAGE_SIZE}`,
|
||||
params,
|
||||
`SELECT p.id, p.url, p.title, t.nickname AS poet, t.url AS poet_url
|
||||
${sql} ORDER BY p.search_text ILIKE $${params.length + 1} DESC, t.birth_year_ah NULLS LAST, p.id
|
||||
LIMIT ${PAGE_SIZE} OFFSET ${(page - 1) * PAGE_SIZE}`,
|
||||
[...params, phrase],
|
||||
),
|
||||
]);
|
||||
return { total: count.rows[0].n, page, pageSize: PAGE_SIZE, results: rows.rows };
|
||||
// snippet: the verse holding most of the terms, best the whole phrase (with its couplet partner), else the first line
|
||||
const ts = terms(req.query.q ?? ''), whole = ts.join(' '), top = ts.length + (ts.length > 1 ? 1 : 0);
|
||||
const verses = await pool.query(
|
||||
'SELECT poem_id, position, couplet, text FROM verses WHERE poem_id = ANY($1) ORDER BY poem_id, vorder',
|
||||
[rows.rows.map((r) => r.id)],
|
||||
);
|
||||
const byPoem = Map.groupBy(verses.rows, (v) => v.poem_id);
|
||||
const results = rows.rows.map(({ id, ...r }) => {
|
||||
const vs = byPoem.get(id) ?? [];
|
||||
let best = vs[0], score = 0;
|
||||
for (const v of vs) {
|
||||
const n = normalise(v.text), k = ts.filter((t) => n.includes(t)).length + (ts.length > 1 && n.includes(whole) ? 1 : 0);
|
||||
if (k > score) [best, score] = [v, k];
|
||||
if (score === top) break;
|
||||
}
|
||||
const lines = !best ? [] : best.position === 'Paragraph' || best.position === 'Single' ? [best.text]
|
||||
: vs.filter((v) => v.couplet === best.couplet && (v.position === 'Right' || v.position === 'Left')).map((v) => v.text);
|
||||
return { ...r, snippet: lines, prose: best?.position === 'Paragraph' };
|
||||
});
|
||||
return { total: count.rows[0].n, page, pageSize: PAGE_SIZE, results };
|
||||
});
|
||||
|
||||
const port = Number(process.env.PORT ?? 4100);
|
||||
|
||||
@ -1,6 +1,6 @@
|
||||
import { test } from 'node:test';
|
||||
import assert from 'node:assert/strict';
|
||||
import { normalise, likePatterns } from './urdu.ts';
|
||||
import { normalise, likePatterns, highlighter } from './urdu.ts';
|
||||
|
||||
test('Urdu variants and diacritics normalise to the same form', () => {
|
||||
const same = (a: string, b: string) => assert.equal(normalise(a), normalise(b), `${a} vs ${b}`);
|
||||
@ -18,3 +18,13 @@ test('search patterns: words, phrases, escaping, empty', () => {
|
||||
assert.deepEqual(likePatterns('50%_x'), ['%50\\%\\_x%']);
|
||||
assert.deepEqual(likePatterns(' '), []);
|
||||
});
|
||||
|
||||
test('highlighter finds terms in original text despite variants and diacritics', () => {
|
||||
const marks = (q: string, text: string) => text.split(highlighter(q)!).filter((_, i) => i % 2);
|
||||
assert.deepEqual(marks('دل ناداں', 'دلِ ناداں تجھے ہوا کیا ہے'), ['دلِ', 'ناداں']);
|
||||
assert.deepEqual(marks('"دل ناداں"', 'دلِ ناداں تجھے'), ['دلِ ناداں']);
|
||||
assert.deepEqual(marks('کوئی', 'كوئي اميد بر نہيں آتی'), ['كوئي']);
|
||||
assert.deepEqual(marks('نالہ', 'نالۂ دل'), ['نالۂ']);
|
||||
assert.deepEqual(marks('کہ', 'کچھ کھ کہ'), ['کہ']); // do-chashmi stays distinct
|
||||
assert.equal(highlighter(' '), null);
|
||||
});
|
||||
|
||||
@ -33,12 +33,29 @@ export function normalise(text: string): string {
|
||||
.trim();
|
||||
}
|
||||
|
||||
// ILIKE patterns for a search term: "quoted phrase" -> one pattern, otherwise one per word (all must match).
|
||||
export function likePatterns(term: string): string[] {
|
||||
// Normalised search terms: "quoted phrase" -> one term, otherwise one per word (all must match).
|
||||
export function terms(term: string): string[] {
|
||||
const t = (term ?? '').trim();
|
||||
if (!t) return [];
|
||||
const phrase = t.length > 1 && t.startsWith('"') && t.endsWith('"');
|
||||
const n = normalise(t);
|
||||
const parts = phrase ? [n] : n.split(' ');
|
||||
return parts.filter(Boolean).map((p) => '%' + p.replace(/[\\%_]/g, (c) => '\\' + c) + '%');
|
||||
return (phrase ? [n] : n.split(' ')).filter(Boolean);
|
||||
}
|
||||
|
||||
// ILIKE patterns, one per term.
|
||||
export const likePatterns = (term: string) => terms(term).map((p) => '%' + p.replace(/[\\%_]/g, (c) => '\\' + c) + '%');
|
||||
|
||||
// A regex that finds the terms in original (unnormalised) text, with one capture group so
|
||||
// text.split(re) puts the matches at odd indexes. Each letter also matches its variants, with
|
||||
// diacritics allowed between letters; a space matches spaces and punctuation.
|
||||
const VARIANTS: Record<string, string> = {};
|
||||
for (const [from, to] of Object.entries(LETTERS)) VARIANTS[to] = (VARIANTS[to] ?? to) + from;
|
||||
const SKIP = '[\u064B-\u0652\u0654\u0670\u0657\u0658\u0615\u200D\u200E\u200F]*';
|
||||
export function highlighter(term: string): RegExp | null {
|
||||
const ts = terms(term);
|
||||
if (!ts.length) return null;
|
||||
const one = (t: string) => [...t].map((c) =>
|
||||
c === ' ' ? '[\\s\u200C.\u060C!\u061F:\u061B;*()"\'\u00AB\u00BB\u06D4]+'
|
||||
: VARIANTS[c] ? `[${VARIANTS[c]}]` : c.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')).join(SKIP) + SKIP;
|
||||
return new RegExp(`(${ts.sort((a, b) => b.length - a.length).map(one).join('|')})`, 'iu');
|
||||
}
|
||||
|
||||
5
web/src/components/Seg.astro
Normal file
5
web/src/components/Seg.astro
Normal file
@ -0,0 +1,5 @@
|
||||
---
|
||||
// renders text segments: [text, 'mark' (search hit) | 'pen' (takhallus) | '']
|
||||
const { s } = Astro.props as { s: [string, string][] };
|
||||
---
|
||||
{s.map(([t, k]) => (k === 'mark' ? <mark>{t}</mark> : k === 'pen' ? <span class="takhallus">{t}</span> : t))}
|
||||
@ -13,3 +13,15 @@ const span = (a: number | null, b: number | null, mark: string) =>
|
||||
// Hijri and Gregorian (عیسوی): "۱۲۹۴ – ۱۳۵۷ ھ · ۱۸۷۷ – ۱۹۳۸ ء"
|
||||
export const years = (p: { birth_year_ah: number | null; death_year_ah: number | null; birth_year_ce?: number | null; death_year_ce?: number | null }) =>
|
||||
[span(p.birth_year_ah, p.death_year_ah, 'ھ'), span(p.birth_year_ce ?? null, p.death_year_ce ?? null, 'ء')].filter(Boolean).join(' · ');
|
||||
|
||||
// search highlighting (same matching rules as the API's search)
|
||||
export { highlighter } from '../../../api/src/urdu.ts';
|
||||
// text split so odd indexes are the matches
|
||||
export const parts = (text: string, re: RegExp | null) => (re ? text.split(re) : [text]);
|
||||
// prose: a stretch of text around the first match
|
||||
export function excerpt(text: string, re: RegExp | null, room = 90) {
|
||||
if (text.length <= room * 2) return text;
|
||||
const i = Math.max(0, re ? text.search(re) : 0);
|
||||
const a = Math.max(0, text.lastIndexOf(' ', Math.max(0, i - room))), b = text.indexOf(' ', i + room);
|
||||
return (a > 0 ? '… ' : '') + text.slice(a, b < 0 ? undefined : b).trim() + (b < 0 ? '' : ' …');
|
||||
}
|
||||
|
||||
@ -1,8 +1,9 @@
|
||||
---
|
||||
// Poet (/p238), category (/p238/shaeri/c641) and poem (/p238/shaeri/c641/sh6556) pages
|
||||
import Base from '../layouts/Base.astro';
|
||||
import Seg from '../components/Seg.astro';
|
||||
import { api } from '../lib/api';
|
||||
import { ud, years } from '../lib/urdu';
|
||||
import { ud, years, highlighter, parts } from '../lib/urdu';
|
||||
|
||||
const url = '/' + (Astro.params.path ?? '');
|
||||
const page = await api(`/api/page?url=${encodeURIComponent(url)}`);
|
||||
@ -22,7 +23,11 @@ if (page.type === 'poem') {
|
||||
}
|
||||
}
|
||||
// pen name (marked ؔ in the text) set apart; split() with a capture group puts matches at odd indexes
|
||||
const verse = (line: string) => line.split(/(\S*\u0614\S*)/);
|
||||
// ?hl= (from search results) marks the searched words on the page; segments: [text, 'mark' | 'pen' | '']
|
||||
const hlq = Astro.url.searchParams.get('hl') ?? '';
|
||||
const re = highlighter(hlq);
|
||||
const show = (line: string) => parts(line, re).flatMap((s, i): [string, string][] =>
|
||||
i % 2 ? [[s, 'mark']] : s.split(/(\S*\u0614\S*)/).map((t, j) => [t, j % 2 ? 'pen' : '']));
|
||||
// contents of a divan grouped by radif letter ("ردیف الف … ی"); ے is filed under ی as in printed divans
|
||||
const LETTER = (l: string) => (l === 'ا' ? 'الف' : l === 'ے' ? 'ی' : l);
|
||||
const groups: { letter: string | null; poems: any[] }[] = [];
|
||||
@ -68,13 +73,18 @@ const title = page.type === 'poem' ? page.poem.title : page.type === 'poet' ? pa
|
||||
|
||||
{page.type === 'poem' && (
|
||||
<>
|
||||
<h1>{page.poem.title}</h1>
|
||||
<h1><Seg s={show(page.poem.title)} /></h1>
|
||||
{re && (
|
||||
<p class="hl-note muted">
|
||||
«{hlq}» نمایاں · <a href={`/search?q=${encodeURIComponent(hlq)}`}>تلاش کے نتائج</a> · <a href={url}>نشان ہٹائیں</a>
|
||||
</p>
|
||||
)}
|
||||
{page.poem.radif != null && <p class="muted radif">{page.poem.radif ? `ردیف: ${page.poem.radif}` : 'غیر مردف'}</p>}
|
||||
<article class="poem">
|
||||
{blocks.map((b) =>
|
||||
b.kind === 'couplet' ? <div class="couplet" data-mark={b.mark}>{b.lines.map((l) => <p>{verse(l).map((s, i) => i % 2 ? <span class="takhallus">{s}</span> : s)}</p>)}</div>
|
||||
: b.kind === 'para' ? <p class="para">{b.lines[0]}</p>
|
||||
: <p class="single">{b.lines[0]}</p>
|
||||
b.kind === 'couplet' ? <div class="couplet" data-mark={b.mark}>{b.lines.map((l) => <p><Seg s={show(l)} /></p>)}</div>
|
||||
: b.kind === 'para' ? <p class="para"><Seg s={show(b.lines[0])} /></p>
|
||||
: <p class="single"><Seg s={show(b.lines[0])} /></p>
|
||||
)}
|
||||
</article>
|
||||
<nav class="poem-nav">
|
||||
@ -86,4 +96,5 @@ const title = page.type === 'poem' ? page.poem.title : page.type === 'poet' ? pa
|
||||
)}
|
||||
</>
|
||||
)}
|
||||
{re && <script is:inline>document.querySelector('.poem mark')?.scrollIntoView({ block: 'center' });</script>}
|
||||
</Base>
|
||||
|
||||
@ -1,12 +1,16 @@
|
||||
---
|
||||
import Base from '../layouts/Base.astro';
|
||||
import Seg from '../components/Seg.astro';
|
||||
import { api } from '../lib/api';
|
||||
import { ud } from '../lib/urdu';
|
||||
import { ud, highlighter, parts, excerpt } from '../lib/urdu';
|
||||
|
||||
const q = Astro.url.searchParams.get('q')?.trim() ?? '';
|
||||
const pageNo = Math.max(1, Number(Astro.url.searchParams.get('page')) || 1);
|
||||
const res = q ? await api(`/api/search?q=${encodeURIComponent(q)}&page=${pageNo}`) : null;
|
||||
const pages = res ? Math.ceil(res.total / res.pageSize) : 0;
|
||||
const re = highlighter(q);
|
||||
const hl = (t: string) => parts(t, re).map((s, i): [string, string] => [s, i % 2 ? 'mark' : '']);
|
||||
const open = (url: string) => `${url}?hl=${encodeURIComponent(q)}`;
|
||||
const link = (n: number) => `/search?q=${encodeURIComponent(q)}&page=${n}`;
|
||||
---
|
||||
<Base title={q ? `تلاش: ${q}` : 'تلاش'} q={q}>
|
||||
@ -17,8 +21,12 @@ const link = (n: number) => `/search?q=${encodeURIComponent(q)}&page=${n}`;
|
||||
<p class="muted">{res.total ? `${ud(res.total)} نتائج` : 'کوئی نتیجہ نہیں ملا۔ کم الفاظ یا کلیدی لفظ آزمائیں۔'}</p>
|
||||
{res.results.map((r: any) => (
|
||||
<div class="result">
|
||||
<a href={r.url}>{r.title}</a> · <a class="muted" href={r.poet_url}>{r.poet}</a>
|
||||
{r.first_line && <div class="first-line">{r.first_line}</div>}
|
||||
<a href={open(r.url)}><Seg s={hl(r.title)} /></a> · <a class="muted" href={r.poet_url}>{r.poet}</a>
|
||||
{r.snippet.length > 0 && (
|
||||
<a class="snippet" href={open(r.url)}>
|
||||
{r.prose ? <p><Seg s={hl(excerpt(r.snippet[0], re))} /></p> : r.snippet.map((l: string) => <p><Seg s={hl(l)} /></p>)}
|
||||
</a>
|
||||
)}
|
||||
</div>
|
||||
))}
|
||||
{pages > 1 && (
|
||||
|
||||
@ -105,7 +105,10 @@ h1::after { content: ''; display: block; width: 140px; height: 2px; background:
|
||||
|
||||
/* search */
|
||||
.result { border-bottom: 1px dashed var(--gold-light); padding: 8px 0; }
|
||||
.result .first-line { color: var(--muted); }
|
||||
.result .snippet { display: block; color: var(--ink); margin-top: 4px; padding: 4px 12px; border-inline-start: 2px solid var(--gold-light); }
|
||||
.result .snippet p { margin: 0; }
|
||||
mark { background: var(--gold-subtle); color: inherit; border-radius: 3px; padding: 0 2px; box-shadow: inset 0 -2px var(--gold); }
|
||||
.hl-note { text-align: center; font-size: .85rem; }
|
||||
.pager { display: flex; gap: 12px; justify-content: center; margin-top: 20px; }
|
||||
|
||||
.site-footer { border-top: 1.5px solid var(--border); padding: 16px 0; font-size: .85rem; color: var(--muted); text-align: center; }
|
||||
|
||||
Loading…
Reference in New Issue
Block a user