Merge pull request 'Divan-owned content: works edited in Divan, plus the content management model' (#53) from feature/divan-owned-content into main

Reviewed-on: #53
This commit is contained in:
anas 2026-10-08 22:34:40 +00:00
commit 835df62aad
4 changed files with 115 additions and 7 deletions

View File

@ -39,7 +39,8 @@ test('prose: chapter heading, paragraphs, footnote', () => {
assert.equal(prose.meta['مصنف'], 'سید احمد خان');
assert.deepEqual(prose.blocks.map((b) => b.type), ['heading', 'para', 'para', 'para']);
assert.equal((prose.blocks[1] as any).line.notes.length, 1);
assert.deepEqual(toVerses(prose).map((v) => v.Position), ['Paragraph', 'Paragraph', 'Paragraph']);
assert.deepEqual(toVerses(prose).map((v) => v.Position), ['Heading', 'Paragraph', 'Paragraph', 'Paragraph']);
assert.equal(toVerses(parse('== باب ==\n\n=== فصل ===\n\nمتن')).map((v) => v.Level ?? 0).join(), '2,3,0', 'chapter and sub-heading levels');
});
test('stanzas and single lines', () => {

View File

@ -17,7 +17,7 @@
export type Inline = { text: string; words: { shown: string; lemma: string }[]; notes: string[]; variants: { shown: string; others: string[]; source?: string }[] };
export type Block =
| { type: 'heading'; text: string }
| { type: 'heading'; text: string; level: number } // == chapter == is level 2, === sub-heading === level 3, …
| { type: 'para'; line: Inline }
| { type: 'couplet' | 'line' | 'stanza'; lines: Inline[]; label?: string };
export type Doc = { meta: Record<string, string>; blocks: Block[] };
@ -76,8 +76,8 @@ export function parse(src: string): Doc {
if (!lines.length) continue;
if (!verse) {
for (const l of lines) {
const h = l.match(/^=+\s*(.+?)\s*=+$/);
blocks.push(h ? { type: 'heading', text: h[1] } : { type: 'para', line: inline(l) });
const h = l.match(/^(=+)\s*(.+?)\s*=+$/);
blocks.push(h ? { type: 'heading', text: h[2], level: h[1].length } : { type: 'para', line: inline(l) });
}
continue;
}
@ -98,10 +98,11 @@ export const words = (text: string) =>
// → the site's verse JSON (the ganjoor-data layout divan-data already exports)
export function toVerses(doc: Doc) {
const out: { VOrder: number; Position: string; Text: string; CoupletIndex: number }[] = [];
const out: { VOrder: number; Position: string; Text: string; CoupletIndex: number; Level?: number }[] = [];
let couplet = 0;
for (const b of doc.blocks) {
if (b.type === 'heading') continue;
// headings inside a work (chapter, sub-heading) are their own kind of verse entry
if (b.type === 'heading') { out.push({ VOrder: 0, Position: 'Heading', Text: b.text, CoupletIndex: couplet++, Level: b.level }); continue; }
if (b.type === 'para') out.push({ VOrder: 0, Position: 'Paragraph', Text: b.line.text, CoupletIndex: couplet++ });
else if (b.type === 'couplet') { b.lines.forEach((l, i) => out.push({ VOrder: 0, Position: i ? 'Left' : 'Right', Text: l.text, CoupletIndex: couplet })); couplet++; }
else for (const l of b.lines) out.push({ VOrder: 0, Position: 'Single', Text: l.text, CoupletIndex: couplet++ });
@ -123,7 +124,7 @@ export function toTEI(doc: Doc) {
const m = doc.meta, body: string[] = [];
let n = 0;
for (const b of doc.blocks) {
if (b.type === 'heading') body.push(`<head>${x(b.text)}</head>`);
if (b.type === 'heading') body.push(`<head type="h${b.level}">${x(b.text)}</head>`);
else if (b.type === 'para') body.push(`<p>${teiLine(b.line)}</p>`);
else {
const type = b.type === 'couplet' ? 'sher' : b.type === 'stanza' ? 'band' : 'misra';

51
api/src/owned.ts Normal file
View File

@ -0,0 +1,51 @@
// Divan-owned content (#30): works edited in Divan are kept in divan-data's divan/ folder, so the daily
// Wikisource sync never overwrites them (export_divan.py prefers them). Each work is two files:
// divan/<work url>.dtx Divan text, the readable source (see divantext.ts, docs/content-model.md)
// divan/<work url>.json generated from it: Title, Verses (the site's verse layout), Edited (who, when)
// Publishing (#34) calls writeOwned and commits both; fromPoem gives the starting text for an edit.
// node src/owned.ts <divan-data dir> <work url> <file.dtx> [editor] write one by hand
import { mkdir, readFile, writeFile } from 'node:fs/promises';
import { dirname, join } from 'node:path';
import { parse, toVerses } from './divantext.ts';
type Verse = { Position: string; Text: string; CoupletIndex: number };
// a work (ganjoor-data poem JSON) as Divan text: verse in <poem> (blank line between couplets, stanzas and
// lines), prose as paragraphs
export function fromPoem(poem: { Title: string; Verses: Verse[]; SourceUrl?: string }, meta: Record<string, string> = {}) {
const head = Object.entries({ عنوان: poem.Title, ...meta, ...(poem.SourceUrl && { ماخذ: 'ویکی ماخذ', ماخذ_ربط: poem.SourceUrl }) })
.map(([k, v]) => `| ${k} = ${v}`).join('\n');
const units = new Map<number, Verse[]>();
for (const v of poem.Verses) (units.get(v.CoupletIndex) ?? units.set(v.CoupletIndex, []).get(v.CoupletIndex)!).push(v);
const parts: string[] = [];
let verse: string[] = [];
const flush = () => { if (verse.length) parts.push(`<poem>\n${verse.join('\n\n')}\n</poem>`); verse = []; };
for (const u of units.values()) {
if (u[0].Position === 'Paragraph') { flush(); parts.push(u.map((v) => v.Text).join('\n')); }
else if (u[0].Position === 'Heading') { flush(); parts.push(`${'='.repeat((u[0] as any).Level ?? 2)} ${u[0].Text} ${'='.repeat((u[0] as any).Level ?? 2)}`); }
else verse.push(u.map((v) => v.Text).join('\n'));
}
flush();
return `{{دیوان\n${head}\n}}\n${parts.join('\n\n')}\n`;
}
// write a Divan-owned work (the .dtx and the generated .json); returns the paths written
export async function writeOwned(dataDir: string, url: string, dtx: string, edited: { by: string; at?: string }) {
const doc = parse(dtx);
const verses = toVerses(doc);
if (!verses.length) throw new Error('the text has no verses or paragraphs');
const base = join(dataDir, 'divan', url.replace(/^\/+|\/+$/g, ''));
if (!/^[\w/-]+$/.test(url) || url.includes('..')) throw new Error(`not a work url: ${url}`);
await mkdir(dirname(base), { recursive: true });
const json = { FullUrl: '/' + url.replace(/^\/+/, ''), Title: doc.meta['عنوان'] ?? '', Verses: verses.map(({ VOrder, ...v }) => ({ VOrder, ...v, SectionIndex1: 0 })),
Edited: { by: edited.by, at: edited.at ?? new Date().toISOString() } };
await writeFile(base + '.dtx', dtx.endsWith('\n') ? dtx : dtx + '\n');
await writeFile(base + '.json', JSON.stringify(json, null, 1) + '\n');
return [base + '.dtx', base + '.json'];
}
if (import.meta.url === `file://${process.argv[1]}`) {
const [dir, url, file, by = 'divan'] = process.argv.slice(2);
if (!dir || !url || !file) { console.error('usage: node src/owned.ts <divan-data dir> <work url> <file.dtx> [editor]'); process.exit(1); }
console.log((await writeOwned(dir, url, await readFile(file, 'utf8'), { by })).join('\n'));
}

View File

@ -115,6 +115,61 @@ above.
8. **Later**: Wikibase as linked-data records for poets and works, if that becomes useful; OpenITI-style
stable identifiers for works and versions.
## Content management model
Owner direction (October 2026): this is **content management**, not only text editing. Every piece of content is
identified for what it is, every change is a version with who did what, and content carries searchable tags.
### What is identified
| Entity | What it is | Its text or fields |
|---|---|---|
| شاعر / ادیب (poet, writer) | A person | Name, pen name, years, places; links to their books |
| تعارف (intro) | A poet's or a book's introduction | Divan text (paragraphs, headings) |
| کتاب (book) | A published collection | Title, year, publisher/edition, order of chapters; intro |
| باب / ذیلی باب (chapter, sub-chapter) | A section of a book, nested | Title, order of works |
| کلام (work) | A ghazal, nazm, rubai, prose chapter, … | Data points (genre, radif, metre, sources) and its text in Divan text |
| Inside a work's text | Identified by Divan text: chapter **heading** (level 2), **sub-heading** (level 3, 4, …), **شعر** (couplet) of two **مصرع** (lines), **بند** (stanza), single line, **paragraph**, **footnote**, **variant reading**, **word** (with its dictionary lemma) | Divan text |
| لغت (dictionary entry) | A word's meanings, readings, pronunciation | Fields |
| ٹیگ (tag) | A typed label | Type and name |
Every entity has a stable id. Elements inside a work (couplets, paragraphs, headings) are addressed by their
position, as the site already does with `#c3` links and bookmarks. The prototype shows the whole library can be
written this way: all 11,087 works convert to Divan text and back unchanged (the only clean-ups are repeated spaces
and raw `== … ==` markers in prose becoming real chapter headings).
### Versions: who did what (#51)
- **Revisions**: every change to any entity is a revision. It records the entity, the version number, the full
content (Divan text or fields), the author, the time, an edit summary and its status. Statuses are draft,
submitted, L1-approved, published, returned and rejected; reviewers and publishers are recorded too.
- **History** page per entity: each version with who, when, summary and status. Any two versions can be compared
with a **diff** (by line for Divan text, by field for metadata). **Revert** creates a new revision through the
pipeline.
- **Moderation log** across the site: who did what, filtered by person, entity, date or action.
- **Pipeline**: the approval pipeline (#31) moves revisions through their statuses. Publishing writes the published
version to divan-data and commits it (#34), so the git history mirrors the published revisions.
### Tags (#52)
- **Types**: tags are typed: موضوع (theme), صنف (genre), بحر (metre), شخصیت (person mentioned), مقام (place),
دور (period), and free tags.
- **Where**: they attach to any entity, down to a couplet or a phrase.
- **Finding tagged content**:
- every tag has a page listing everything tagged with it;
- tags are a search filter and facet, like authors;
- `tag:تصوف` can be typed in the search box.
- **History and permissions**: adding or removing a tag is a revision like any other change, so its history shows
who tagged what and when. The permission model gets a `tags` content type.
### Storage
- **PostgreSQL** holds the entities (`poets`, `categories`, `poems`, `verses` today), plus `revisions`, `tags` and
`entity_tags`.
- **divan-data** holds the published versions: works as `.dtx` with generated `.json` in `divan/` (#30, done in
this step). Intros, books and chapters follow the same pattern as they become editable, and tags go in the
Divan text header (`| ٹیگ = عشق، تصوف`) and a tags list.
## Prototype results
`api/src/divantext.ts` parses Divan text into blocks, converts them to the site's verse JSON, and exports TEI.