From 8d825277654022f31d4da2066d957ccf51139473 Mon Sep 17 00:00:00 2001 From: Anas Rashid Date: Fri, 9 Oct 2026 00:22:31 +0200 Subject: [PATCH] Divan-owned content (#30): write works as Divan text + generated JSON for divan-data; headings; content management model MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - api/src/owned.ts: fromPoem (a work as Divan text, the start of an edit) and writeOwned (the .dtx source and the generated .json into divan-data/divan/); a CLI for writing one by hand. - Divan text: chapter headings and sub-headings with levels, as their own verse entries. - All 11,087 works convert to Divan text and back unchanged (apart from repeated spaces, and raw '== … ==' markers in prose becoming headings). - docs/content-model.md: the content management model (what is identified, versions and who did what (#51), tags and tag search (#52), storage). Co-Authored-By: Claude Opus 5.5 --- api/src/divantext.test.ts | 3 ++- api/src/divantext.ts | 13 ++++----- api/src/owned.ts | 51 ++++++++++++++++++++++++++++++++++++ docs/content-model.md | 55 +++++++++++++++++++++++++++++++++++++++ 4 files changed, 115 insertions(+), 7 deletions(-) create mode 100644 api/src/owned.ts diff --git a/api/src/divantext.test.ts b/api/src/divantext.test.ts index 836f9c1b..9e7bed4e 100644 --- a/api/src/divantext.test.ts +++ b/api/src/divantext.test.ts @@ -39,7 +39,8 @@ test('prose: chapter heading, paragraphs, footnote', () => { assert.equal(prose.meta['مصنف'], 'سید احمد خان'); assert.deepEqual(prose.blocks.map((b) => b.type), ['heading', 'para', 'para', 'para']); assert.equal((prose.blocks[1] as any).line.notes.length, 1); - assert.deepEqual(toVerses(prose).map((v) => v.Position), ['Paragraph', 'Paragraph', 'Paragraph']); + assert.deepEqual(toVerses(prose).map((v) => v.Position), ['Heading', 'Paragraph', 'Paragraph', 'Paragraph']); + assert.equal(toVerses(parse('== باب ==\n\n=== فصل ===\n\nمتن')).map((v) => v.Level ?? 0).join(), '2,3,0', 'chapter and sub-heading levels'); }); test('stanzas and single lines', () => { diff --git a/api/src/divantext.ts b/api/src/divantext.ts index fa0ae53c..1051eee6 100644 --- a/api/src/divantext.ts +++ b/api/src/divantext.ts @@ -17,7 +17,7 @@ export type Inline = { text: string; words: { shown: string; lemma: string }[]; notes: string[]; variants: { shown: string; others: string[]; source?: string }[] }; export type Block = - | { type: 'heading'; text: string } + | { type: 'heading'; text: string; level: number } // == chapter == is level 2, === sub-heading === level 3, … | { type: 'para'; line: Inline } | { type: 'couplet' | 'line' | 'stanza'; lines: Inline[]; label?: string }; export type Doc = { meta: Record; blocks: Block[] }; @@ -76,8 +76,8 @@ export function parse(src: string): Doc { if (!lines.length) continue; if (!verse) { for (const l of lines) { - const h = l.match(/^=+\s*(.+?)\s*=+$/); - blocks.push(h ? { type: 'heading', text: h[1] } : { type: 'para', line: inline(l) }); + const h = l.match(/^(=+)\s*(.+?)\s*=+$/); + blocks.push(h ? { type: 'heading', text: h[2], level: h[1].length } : { type: 'para', line: inline(l) }); } continue; } @@ -98,10 +98,11 @@ export const words = (text: string) => // → the site's verse JSON (the ganjoor-data layout divan-data already exports) export function toVerses(doc: Doc) { - const out: { VOrder: number; Position: string; Text: string; CoupletIndex: number }[] = []; + const out: { VOrder: number; Position: string; Text: string; CoupletIndex: number; Level?: number }[] = []; let couplet = 0; for (const b of doc.blocks) { - if (b.type === 'heading') continue; + // headings inside a work (chapter, sub-heading) are their own kind of verse entry + if (b.type === 'heading') { out.push({ VOrder: 0, Position: 'Heading', Text: b.text, CoupletIndex: couplet++, Level: b.level }); continue; } if (b.type === 'para') out.push({ VOrder: 0, Position: 'Paragraph', Text: b.line.text, CoupletIndex: couplet++ }); else if (b.type === 'couplet') { b.lines.forEach((l, i) => out.push({ VOrder: 0, Position: i ? 'Left' : 'Right', Text: l.text, CoupletIndex: couplet })); couplet++; } else for (const l of b.lines) out.push({ VOrder: 0, Position: 'Single', Text: l.text, CoupletIndex: couplet++ }); @@ -123,7 +124,7 @@ export function toTEI(doc: Doc) { const m = doc.meta, body: string[] = []; let n = 0; for (const b of doc.blocks) { - if (b.type === 'heading') body.push(`${x(b.text)}`); + if (b.type === 'heading') body.push(`${x(b.text)}`); else if (b.type === 'para') body.push(`

${teiLine(b.line)}

`); else { const type = b.type === 'couplet' ? 'sher' : b.type === 'stanza' ? 'band' : 'misra'; diff --git a/api/src/owned.ts b/api/src/owned.ts new file mode 100644 index 00000000..4b93aab6 --- /dev/null +++ b/api/src/owned.ts @@ -0,0 +1,51 @@ +// Divan-owned content (#30): works edited in Divan are kept in divan-data's divan/ folder, so the daily +// Wikisource sync never overwrites them (export_divan.py prefers them). Each work is two files: +// divan/.dtx Divan text, the readable source (see divantext.ts, docs/content-model.md) +// divan/.json generated from it: Title, Verses (the site's verse layout), Edited (who, when) +// Publishing (#34) calls writeOwned and commits both; fromPoem gives the starting text for an edit. +// node src/owned.ts [editor] write one by hand +import { mkdir, readFile, writeFile } from 'node:fs/promises'; +import { dirname, join } from 'node:path'; +import { parse, toVerses } from './divantext.ts'; + +type Verse = { Position: string; Text: string; CoupletIndex: number }; + +// a work (ganjoor-data poem JSON) as Divan text: verse in (blank line between couplets, stanzas and +// lines), prose as paragraphs +export function fromPoem(poem: { Title: string; Verses: Verse[]; SourceUrl?: string }, meta: Record = {}) { + const head = Object.entries({ عنوان: poem.Title, ...meta, ...(poem.SourceUrl && { ماخذ: 'ویکی ماخذ', ماخذ_ربط: poem.SourceUrl }) }) + .map(([k, v]) => `| ${k} = ${v}`).join('\n'); + const units = new Map(); + for (const v of poem.Verses) (units.get(v.CoupletIndex) ?? units.set(v.CoupletIndex, []).get(v.CoupletIndex)!).push(v); + const parts: string[] = []; + let verse: string[] = []; + const flush = () => { if (verse.length) parts.push(`\n${verse.join('\n\n')}\n`); verse = []; }; + for (const u of units.values()) { + if (u[0].Position === 'Paragraph') { flush(); parts.push(u.map((v) => v.Text).join('\n')); } + else if (u[0].Position === 'Heading') { flush(); parts.push(`${'='.repeat((u[0] as any).Level ?? 2)} ${u[0].Text} ${'='.repeat((u[0] as any).Level ?? 2)}`); } + else verse.push(u.map((v) => v.Text).join('\n')); + } + flush(); + return `{{دیوان\n${head}\n}}\n${parts.join('\n\n')}\n`; +} + +// write a Divan-owned work (the .dtx and the generated .json); returns the paths written +export async function writeOwned(dataDir: string, url: string, dtx: string, edited: { by: string; at?: string }) { + const doc = parse(dtx); + const verses = toVerses(doc); + if (!verses.length) throw new Error('the text has no verses or paragraphs'); + const base = join(dataDir, 'divan', url.replace(/^\/+|\/+$/g, '')); + if (!/^[\w/-]+$/.test(url) || url.includes('..')) throw new Error(`not a work url: ${url}`); + await mkdir(dirname(base), { recursive: true }); + const json = { FullUrl: '/' + url.replace(/^\/+/, ''), Title: doc.meta['عنوان'] ?? '', Verses: verses.map(({ VOrder, ...v }) => ({ VOrder, ...v, SectionIndex1: 0 })), + Edited: { by: edited.by, at: edited.at ?? new Date().toISOString() } }; + await writeFile(base + '.dtx', dtx.endsWith('\n') ? dtx : dtx + '\n'); + await writeFile(base + '.json', JSON.stringify(json, null, 1) + '\n'); + return [base + '.dtx', base + '.json']; +} + +if (import.meta.url === `file://${process.argv[1]}`) { + const [dir, url, file, by = 'divan'] = process.argv.slice(2); + if (!dir || !url || !file) { console.error('usage: node src/owned.ts [editor]'); process.exit(1); } + console.log((await writeOwned(dir, url, await readFile(file, 'utf8'), { by })).join('\n')); +} diff --git a/docs/content-model.md b/docs/content-model.md index 632e53d6..83d225b3 100644 --- a/docs/content-model.md +++ b/docs/content-model.md @@ -115,6 +115,61 @@ above. 8. **Later**: Wikibase as linked-data records for poets and works, if that becomes useful; OpenITI-style stable identifiers for works and versions. +## Content management model + +Owner direction (October 2026): this is **content management**, not only text editing. Every piece of content is +identified for what it is, every change is a version with who did what, and content carries searchable tags. + +### What is identified + +| Entity | What it is | Its text or fields | +|---|---|---| +| شاعر / ادیب (poet, writer) | A person | Name, pen name, years, places; links to their books | +| تعارف (intro) | A poet's or a book's introduction | Divan text (paragraphs, headings) | +| کتاب (book) | A published collection | Title, year, publisher/edition, order of chapters; intro | +| باب / ذیلی باب (chapter, sub-chapter) | A section of a book, nested | Title, order of works | +| کلام (work) | A ghazal, nazm, rubai, prose chapter, … | Data points (genre, radif, metre, sources) and its text in Divan text | +| Inside a work's text | Identified by Divan text: chapter **heading** (level 2), **sub-heading** (level 3, 4, …), **شعر** (couplet) of two **مصرع** (lines), **بند** (stanza), single line, **paragraph**, **footnote**, **variant reading**, **word** (with its dictionary lemma) | Divan text | +| لغت (dictionary entry) | A word's meanings, readings, pronunciation | Fields | +| ٹیگ (tag) | A typed label | Type and name | + +Every entity has a stable id. Elements inside a work (couplets, paragraphs, headings) are addressed by their +position, as the site already does with `#c3` links and bookmarks. The prototype shows the whole library can be +written this way: all 11,087 works convert to Divan text and back unchanged (the only clean-ups are repeated spaces +and raw `== … ==` markers in prose becoming real chapter headings). + +### Versions: who did what (#51) + +- **Revisions**: every change to any entity is a revision. It records the entity, the version number, the full + content (Divan text or fields), the author, the time, an edit summary and its status. Statuses are draft, + submitted, L1-approved, published, returned and rejected; reviewers and publishers are recorded too. +- **History** page per entity: each version with who, when, summary and status. Any two versions can be compared + with a **diff** (by line for Divan text, by field for metadata). **Revert** creates a new revision through the + pipeline. +- **Moderation log** across the site: who did what, filtered by person, entity, date or action. +- **Pipeline**: the approval pipeline (#31) moves revisions through their statuses. Publishing writes the published + version to divan-data and commits it (#34), so the git history mirrors the published revisions. + +### Tags (#52) + +- **Types**: tags are typed: موضوع (theme), صنف (genre), بحر (metre), شخصیت (person mentioned), مقام (place), + دور (period), and free tags. +- **Where**: they attach to any entity, down to a couplet or a phrase. +- **Finding tagged content**: + - every tag has a page listing everything tagged with it; + - tags are a search filter and facet, like authors; + - `tag:تصوف` can be typed in the search box. +- **History and permissions**: adding or removing a tag is a revision like any other change, so its history shows + who tagged what and when. The permission model gets a `tags` content type. + +### Storage + +- **PostgreSQL** holds the entities (`poets`, `categories`, `poems`, `verses` today), plus `revisions`, `tags` and + `entity_tags`. +- **divan-data** holds the published versions: works as `.dtx` with generated `.json` in `divan/` (#30, done in + this step). Intros, books and chapters follow the same pattern as they become editable, and tags go in the + Divan text header (`| ٹیگ = عشق، تصوف`) and a tags list. + ## Prototype results `api/src/divantext.ts` parses Divan text into blocks, converts them to the site's verse JSON, and exports TEI.