diff --git a/README.md b/README.md index 73448b71..3446011c 100644 --- a/README.md +++ b/README.md @@ -54,7 +54,7 @@ Settings: `DIVAN_DATA_DIR` (default `/opt/divan-data`, cloned on first run), `DI |---|---| | `GET /api/poets` | all poets | | `GET /api/page?url=/p238/...` | the poet, category or poem at a site URL (breadcrumbs, children, verses, prev/next) | -| `GET /api/search?q=&poet=&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised; exact phrase first; each with the best-matching couplet or paragraph (`snippet`) | +| `GET /api/search?q=&poet=1,2&page=` | poems containing all words (or a `"quoted phrase"`), Urdu-normalised; exact phrase first; each with the best-matching couplet or paragraph (`snippet`); optionally only some poets/writers; plus `authors` (who the results come from, with counts), and on page 1 `poets` (by name) and `books` (books/chapters by title) | | `GET /api/word?w=` | one word's meanings and pronunciation from the local Wiktionary data (Urdu, Persian, Arabic in that order; English meanings; Urdu equivalents via English when Urdu Wiktionary has none) | | `/api/auth/*` | accounts: sign-up, sign-in (returns a Bearer token), profile, password, delete | | `/api/library/*` | the signed-in reader's library: toggle poets, works, couplets and words; list with full paths; notes | diff --git a/api/src/search.test.ts b/api/src/search.test.ts new file mode 100644 index 00000000..2439c807 --- /dev/null +++ b/api/src/search.test.ts @@ -0,0 +1,20 @@ +import { test, after } from 'node:test'; +import assert from 'node:assert/strict'; +import { nameMatches } from './search.ts'; +import { terms } from './urdu.ts'; +import { pool } from './db.ts'; + +after(() => pool.end()); + +test('poets/writers by name or pen name; books/chapters by title; Urdu variants match', async () => { + const ghalib = await nameMatches(terms('غالب')); + assert.ok(ghalib.poets.some((p) => p.url === '/p266'), 'Ghalib by pen name'); + const iqbal = await nameMatches(terms('اقبال')); + assert.equal(iqbal.poets[0].url, '/p238', 'most works first'); + const book = await nameMatches(terms('بانگ درا')); + assert.ok(book.books.some((b) => b.url.startsWith('/p238/')), 'Iqbal\'s book by title'); + assert.ok(book.books[0].poet && book.books[0].works > 0); + const arabic = await nameMatches(terms('مير تقي مير')); // Arabic yeh + assert.ok(arabic.poets.some((p) => p.url === '/p303'), 'Arabic ي/ی variants'); + assert.deepEqual(await nameMatches(terms('ژژژژژ')), { poets: [], books: [] }); +}); diff --git a/api/src/search.ts b/api/src/search.ts new file mode 100644 index 00000000..8616b633 --- /dev/null +++ b/api/src/search.ts @@ -0,0 +1,19 @@ +import { pool } from './db.ts'; +import { normalise } from './urdu.ts'; + +// poets/writers by name or pen name, and books/chapters by title, matched with the same Urdu normalisation +// as the content; ponytail: ~1,400 rows filtered in memory per search, add normalised columns + trigram index if it grows +export async function nameMatches(terms: string[]) { + const has = (text: string) => { const n = normalise(text); return terms.every((t) => n.includes(t)); }; + const [poets, cats] = await Promise.all([ + pool.query(`SELECT p.id, p.url, p.name, p.nickname, p.birth_year_ah, p.death_year_ah, p.birth_year_ce, p.death_year_ce, + (SELECT count(*)::int FROM poems w WHERE w.poet_id = p.id) AS works FROM poets p`), + pool.query(`SELECT c.id, c.url, c.title, t.nickname AS poet, t.url AS poet_url, + (SELECT count(*)::int FROM poems w WHERE w.category_id = c.id) AS works + FROM categories c JOIN poets t ON t.id = c.poet_id WHERE c.parent_id IS NOT NULL`), + ]); + return { + poets: poets.rows.filter((p) => has(`${p.name} ${p.nickname}`)).sort((a, b) => b.works - a.works).slice(0, 24), + books: cats.rows.filter((c) => has(c.title)).sort((a, b) => b.works - a.works).slice(0, 12), + }; +} diff --git a/api/src/server.ts b/api/src/server.ts index 50212271..65b20667 100644 --- a/api/src/server.ts +++ b/api/src/server.ts @@ -1,7 +1,8 @@ // Divan API (Fastify + PostgreSQL) // GET /api/poets all poets // GET /api/page?url=/p238/... poet, category or poem at that URL -// GET /api/search?q=&poet=&page= +// GET /api/search?q=&poet=1,2&page= content results (paged, optionally only these poets/writers), the poets/writers +// the results come from (with counts), poets by name and books/chapters by title // GET /api/word?w= Wiktionary meanings and pronunciation (sidebar) // /api/auth/* accounts (see auth.ts) // /api/admin/* admin panel (see admin.ts) @@ -10,6 +11,7 @@ import Fastify from 'fastify'; import { pool } from './db.ts'; import { likePatterns, normalise, terms } from './urdu.ts'; +import { nameMatches } from './search.ts'; import { lookup, PUNCT } from './dictionary.ts'; import { authRoutes } from './auth.ts'; import { adminRoutes } from './admin.ts'; @@ -91,13 +93,15 @@ app.get<{ Querystring: { url?: string } }>('/api/page', async (req, reply) => { app.get<{ Querystring: { q?: string; poet?: string; page?: string } }>('/api/search', async (req) => { const patterns = likePatterns(req.query.q ?? ''); - if (!patterns.length) return { total: 0, page: 1, results: [] }; + if (!patterns.length) return { total: 0, page: 1, pageSize: PAGE_SIZE, results: [], poets: [], books: [] }; const page = Math.max(1, Number(req.query.page) || 1); const params: unknown[] = [...patterns]; const where = patterns.map((_, i) => `p.search_text ILIKE $${i + 1}`); - if (req.query.poet) { - params.push(Number(req.query.poet)); - where.push(`p.poet_id = $${params.length}`); + const textWhere = where.join(' AND '); + const poetIds = [...new Set(String(req.query.poet ?? '').split(',').map(Number).filter((n) => Number.isInteger(n) && n > 0))].slice(0, 50); + if (poetIds.length) { + params.push(poetIds); + where.push(`p.poet_id = ANY($${params.length})`); } const sql = `FROM poems p JOIN poets t ON t.id = p.poet_id WHERE ${where.join(' AND ')}`; // several words: poems with them together as a phrase come first @@ -130,7 +134,15 @@ app.get<{ Querystring: { q?: string; poet?: string; page?: string } }>('/api/sea : vs.filter((v) => v.couplet === best.couplet && (v.position === 'Right' || v.position === 'Left')).map((v) => v.text); return { ...r, snippet: lines, prose: best?.position === 'Paragraph' }; }); - return { total: count.rows[0].n, page, pageSize: PAGE_SIZE, results }; + // names and titles on the first page: poets/writers whose name has every word, books/chapters likewise + const names = page === 1 && !poetIds.length ? await nameMatches(ts) : { poets: [], books: [] }; + // which poets/writers the matching content comes from (whatever the poet filter), for narrowing down + const [authors, selected] = await Promise.all([ + pool.query(`SELECT t.id, t.url, t.nickname, count(*)::int AS n FROM poems p JOIN poets t ON t.id = p.poet_id + WHERE ${textWhere} GROUP BY t.id ORDER BY n DESC, t.nickname LIMIT 40`, patterns), + poetIds.length ? pool.query('SELECT id, url, nickname FROM poets WHERE id = ANY($1) ORDER BY nickname', [poetIds]) : { rows: [] }, + ]); + return { total: count.rows[0].n, page, pageSize: PAGE_SIZE, results, ...names, authors: authors.rows, selected: selected.rows }; }); // one word in Arabic script (Urdu, Persian, Arabic), as selected by a reader diff --git a/web/src/layouts/Base.astro b/web/src/layouts/Base.astro index 0494a449..a47f25fb 100644 --- a/web/src/layouts/Base.astro +++ b/web/src/layouts/Base.astro @@ -34,7 +34,7 @@ const fullTitle = title ? `${title} · دیوان` : 'دیوان · اردو ک