Divan-owned content (#30): write works as Divan text + generated JSON for divan-data; headings; content management model
- api/src/owned.ts: fromPoem (a work as Divan text, the start of an edit) and writeOwned (the .dtx source and the generated .json into divan-data/divan/); a CLI for writing one by hand. - Divan text: chapter headings and sub-headings with levels, as their own verse entries. - All 11,087 works convert to Divan text and back unchanged (apart from repeated spaces, and raw '== … ==' markers in prose becoming headings). - docs/content-model.md: the content management model (what is identified, versions and who did what (#51), tags and tag search (#52), storage). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
a9626b4d9e
commit
8d82527765
@ -39,7 +39,8 @@ test('prose: chapter heading, paragraphs, footnote', () => {
|
||||
assert.equal(prose.meta['مصنف'], 'سید احمد خان');
|
||||
assert.deepEqual(prose.blocks.map((b) => b.type), ['heading', 'para', 'para', 'para']);
|
||||
assert.equal((prose.blocks[1] as any).line.notes.length, 1);
|
||||
assert.deepEqual(toVerses(prose).map((v) => v.Position), ['Paragraph', 'Paragraph', 'Paragraph']);
|
||||
assert.deepEqual(toVerses(prose).map((v) => v.Position), ['Heading', 'Paragraph', 'Paragraph', 'Paragraph']);
|
||||
assert.equal(toVerses(parse('== باب ==\n\n=== فصل ===\n\nمتن')).map((v) => v.Level ?? 0).join(), '2,3,0', 'chapter and sub-heading levels');
|
||||
});
|
||||
|
||||
test('stanzas and single lines', () => {
|
||||
|
||||
@ -17,7 +17,7 @@
|
||||
|
||||
export type Inline = { text: string; words: { shown: string; lemma: string }[]; notes: string[]; variants: { shown: string; others: string[]; source?: string }[] };
|
||||
export type Block =
|
||||
| { type: 'heading'; text: string }
|
||||
| { type: 'heading'; text: string; level: number } // == chapter == is level 2, === sub-heading === level 3, …
|
||||
| { type: 'para'; line: Inline }
|
||||
| { type: 'couplet' | 'line' | 'stanza'; lines: Inline[]; label?: string };
|
||||
export type Doc = { meta: Record<string, string>; blocks: Block[] };
|
||||
@ -76,8 +76,8 @@ export function parse(src: string): Doc {
|
||||
if (!lines.length) continue;
|
||||
if (!verse) {
|
||||
for (const l of lines) {
|
||||
const h = l.match(/^=+\s*(.+?)\s*=+$/);
|
||||
blocks.push(h ? { type: 'heading', text: h[1] } : { type: 'para', line: inline(l) });
|
||||
const h = l.match(/^(=+)\s*(.+?)\s*=+$/);
|
||||
blocks.push(h ? { type: 'heading', text: h[2], level: h[1].length } : { type: 'para', line: inline(l) });
|
||||
}
|
||||
continue;
|
||||
}
|
||||
@ -98,10 +98,11 @@ export const words = (text: string) =>
|
||||
|
||||
// → the site's verse JSON (the ganjoor-data layout divan-data already exports)
|
||||
export function toVerses(doc: Doc) {
|
||||
const out: { VOrder: number; Position: string; Text: string; CoupletIndex: number }[] = [];
|
||||
const out: { VOrder: number; Position: string; Text: string; CoupletIndex: number; Level?: number }[] = [];
|
||||
let couplet = 0;
|
||||
for (const b of doc.blocks) {
|
||||
if (b.type === 'heading') continue;
|
||||
// headings inside a work (chapter, sub-heading) are their own kind of verse entry
|
||||
if (b.type === 'heading') { out.push({ VOrder: 0, Position: 'Heading', Text: b.text, CoupletIndex: couplet++, Level: b.level }); continue; }
|
||||
if (b.type === 'para') out.push({ VOrder: 0, Position: 'Paragraph', Text: b.line.text, CoupletIndex: couplet++ });
|
||||
else if (b.type === 'couplet') { b.lines.forEach((l, i) => out.push({ VOrder: 0, Position: i ? 'Left' : 'Right', Text: l.text, CoupletIndex: couplet })); couplet++; }
|
||||
else for (const l of b.lines) out.push({ VOrder: 0, Position: 'Single', Text: l.text, CoupletIndex: couplet++ });
|
||||
@ -123,7 +124,7 @@ export function toTEI(doc: Doc) {
|
||||
const m = doc.meta, body: string[] = [];
|
||||
let n = 0;
|
||||
for (const b of doc.blocks) {
|
||||
if (b.type === 'heading') body.push(`<head>${x(b.text)}</head>`);
|
||||
if (b.type === 'heading') body.push(`<head type="h${b.level}">${x(b.text)}</head>`);
|
||||
else if (b.type === 'para') body.push(`<p>${teiLine(b.line)}</p>`);
|
||||
else {
|
||||
const type = b.type === 'couplet' ? 'sher' : b.type === 'stanza' ? 'band' : 'misra';
|
||||
|
||||
51
api/src/owned.ts
Normal file
51
api/src/owned.ts
Normal file
@ -0,0 +1,51 @@
|
||||
// Divan-owned content (#30): works edited in Divan are kept in divan-data's divan/ folder, so the daily
|
||||
// Wikisource sync never overwrites them (export_divan.py prefers them). Each work is two files:
|
||||
// divan/<work url>.dtx Divan text, the readable source (see divantext.ts, docs/content-model.md)
|
||||
// divan/<work url>.json generated from it: Title, Verses (the site's verse layout), Edited (who, when)
|
||||
// Publishing (#34) calls writeOwned and commits both; fromPoem gives the starting text for an edit.
|
||||
// node src/owned.ts <divan-data dir> <work url> <file.dtx> [editor] write one by hand
|
||||
import { mkdir, readFile, writeFile } from 'node:fs/promises';
|
||||
import { dirname, join } from 'node:path';
|
||||
import { parse, toVerses } from './divantext.ts';
|
||||
|
||||
type Verse = { Position: string; Text: string; CoupletIndex: number };
|
||||
|
||||
// a work (ganjoor-data poem JSON) as Divan text: verse in <poem> (blank line between couplets, stanzas and
|
||||
// lines), prose as paragraphs
|
||||
export function fromPoem(poem: { Title: string; Verses: Verse[]; SourceUrl?: string }, meta: Record<string, string> = {}) {
|
||||
const head = Object.entries({ عنوان: poem.Title, ...meta, ...(poem.SourceUrl && { ماخذ: 'ویکی ماخذ', ماخذ_ربط: poem.SourceUrl }) })
|
||||
.map(([k, v]) => `| ${k} = ${v}`).join('\n');
|
||||
const units = new Map<number, Verse[]>();
|
||||
for (const v of poem.Verses) (units.get(v.CoupletIndex) ?? units.set(v.CoupletIndex, []).get(v.CoupletIndex)!).push(v);
|
||||
const parts: string[] = [];
|
||||
let verse: string[] = [];
|
||||
const flush = () => { if (verse.length) parts.push(`<poem>\n${verse.join('\n\n')}\n</poem>`); verse = []; };
|
||||
for (const u of units.values()) {
|
||||
if (u[0].Position === 'Paragraph') { flush(); parts.push(u.map((v) => v.Text).join('\n')); }
|
||||
else if (u[0].Position === 'Heading') { flush(); parts.push(`${'='.repeat((u[0] as any).Level ?? 2)} ${u[0].Text} ${'='.repeat((u[0] as any).Level ?? 2)}`); }
|
||||
else verse.push(u.map((v) => v.Text).join('\n'));
|
||||
}
|
||||
flush();
|
||||
return `{{دیوان\n${head}\n}}\n${parts.join('\n\n')}\n`;
|
||||
}
|
||||
|
||||
// write a Divan-owned work (the .dtx and the generated .json); returns the paths written
|
||||
export async function writeOwned(dataDir: string, url: string, dtx: string, edited: { by: string; at?: string }) {
|
||||
const doc = parse(dtx);
|
||||
const verses = toVerses(doc);
|
||||
if (!verses.length) throw new Error('the text has no verses or paragraphs');
|
||||
const base = join(dataDir, 'divan', url.replace(/^\/+|\/+$/g, ''));
|
||||
if (!/^[\w/-]+$/.test(url) || url.includes('..')) throw new Error(`not a work url: ${url}`);
|
||||
await mkdir(dirname(base), { recursive: true });
|
||||
const json = { FullUrl: '/' + url.replace(/^\/+/, ''), Title: doc.meta['عنوان'] ?? '', Verses: verses.map(({ VOrder, ...v }) => ({ VOrder, ...v, SectionIndex1: 0 })),
|
||||
Edited: { by: edited.by, at: edited.at ?? new Date().toISOString() } };
|
||||
await writeFile(base + '.dtx', dtx.endsWith('\n') ? dtx : dtx + '\n');
|
||||
await writeFile(base + '.json', JSON.stringify(json, null, 1) + '\n');
|
||||
return [base + '.dtx', base + '.json'];
|
||||
}
|
||||
|
||||
if (import.meta.url === `file://${process.argv[1]}`) {
|
||||
const [dir, url, file, by = 'divan'] = process.argv.slice(2);
|
||||
if (!dir || !url || !file) { console.error('usage: node src/owned.ts <divan-data dir> <work url> <file.dtx> [editor]'); process.exit(1); }
|
||||
console.log((await writeOwned(dir, url, await readFile(file, 'utf8'), { by })).join('\n'));
|
||||
}
|
||||
@ -115,6 +115,61 @@ above.
|
||||
8. **Later**: Wikibase as linked-data records for poets and works, if that becomes useful; OpenITI-style
|
||||
stable identifiers for works and versions.
|
||||
|
||||
## Content management model
|
||||
|
||||
Owner direction (October 2026): this is **content management**, not only text editing. Every piece of content is
|
||||
identified for what it is, every change is a version with who did what, and content carries searchable tags.
|
||||
|
||||
### What is identified
|
||||
|
||||
| Entity | What it is | Its text or fields |
|
||||
|---|---|---|
|
||||
| شاعر / ادیب (poet, writer) | A person | Name, pen name, years, places; links to their books |
|
||||
| تعارف (intro) | A poet's or a book's introduction | Divan text (paragraphs, headings) |
|
||||
| کتاب (book) | A published collection | Title, year, publisher/edition, order of chapters; intro |
|
||||
| باب / ذیلی باب (chapter, sub-chapter) | A section of a book, nested | Title, order of works |
|
||||
| کلام (work) | A ghazal, nazm, rubai, prose chapter, … | Data points (genre, radif, metre, sources) and its text in Divan text |
|
||||
| Inside a work's text | Identified by Divan text: chapter **heading** (level 2), **sub-heading** (level 3, 4, …), **شعر** (couplet) of two **مصرع** (lines), **بند** (stanza), single line, **paragraph**, **footnote**, **variant reading**, **word** (with its dictionary lemma) | Divan text |
|
||||
| لغت (dictionary entry) | A word's meanings, readings, pronunciation | Fields |
|
||||
| ٹیگ (tag) | A typed label | Type and name |
|
||||
|
||||
Every entity has a stable id. Elements inside a work (couplets, paragraphs, headings) are addressed by their
|
||||
position, as the site already does with `#c3` links and bookmarks. The prototype shows the whole library can be
|
||||
written this way: all 11,087 works convert to Divan text and back unchanged (the only clean-ups are repeated spaces
|
||||
and raw `== … ==` markers in prose becoming real chapter headings).
|
||||
|
||||
### Versions: who did what (#51)
|
||||
|
||||
- **Revisions**: every change to any entity is a revision. It records the entity, the version number, the full
|
||||
content (Divan text or fields), the author, the time, an edit summary and its status. Statuses are draft,
|
||||
submitted, L1-approved, published, returned and rejected; reviewers and publishers are recorded too.
|
||||
- **History** page per entity: each version with who, when, summary and status. Any two versions can be compared
|
||||
with a **diff** (by line for Divan text, by field for metadata). **Revert** creates a new revision through the
|
||||
pipeline.
|
||||
- **Moderation log** across the site: who did what, filtered by person, entity, date or action.
|
||||
- **Pipeline**: the approval pipeline (#31) moves revisions through their statuses. Publishing writes the published
|
||||
version to divan-data and commits it (#34), so the git history mirrors the published revisions.
|
||||
|
||||
### Tags (#52)
|
||||
|
||||
- **Types**: tags are typed: موضوع (theme), صنف (genre), بحر (metre), شخصیت (person mentioned), مقام (place),
|
||||
دور (period), and free tags.
|
||||
- **Where**: they attach to any entity, down to a couplet or a phrase.
|
||||
- **Finding tagged content**:
|
||||
- every tag has a page listing everything tagged with it;
|
||||
- tags are a search filter and facet, like authors;
|
||||
- `tag:تصوف` can be typed in the search box.
|
||||
- **History and permissions**: adding or removing a tag is a revision like any other change, so its history shows
|
||||
who tagged what and when. The permission model gets a `tags` content type.
|
||||
|
||||
### Storage
|
||||
|
||||
- **PostgreSQL** holds the entities (`poets`, `categories`, `poems`, `verses` today), plus `revisions`, `tags` and
|
||||
`entity_tags`.
|
||||
- **divan-data** holds the published versions: works as `.dtx` with generated `.json` in `divan/` (#30, done in
|
||||
this step). Intros, books and chapters follow the same pattern as they become editable, and tags go in the
|
||||
Divan text header (`| ٹیگ = عشق، تصوف`) and a tags list.
|
||||
|
||||
## Prototype results
|
||||
|
||||
`api/src/divantext.ts` parses Divan text into blocks, converts them to the site's verse JSON, and exports TEI.
|
||||
|
||||
Loading…
Reference in New Issue
Block a user