Divan-owned content (#30): write works as Divan text + generated JSON for divan-data; headings; content management model
- api/src/owned.ts: fromPoem (a work as Divan text, the start of an edit) and writeOwned (the .dtx source and the generated .json into divan-data/divan/); a CLI for writing one by hand. - Divan text: chapter headings and sub-headings with levels, as their own verse entries. - All 11,087 works convert to Divan text and back unchanged (apart from repeated spaces, and raw '== … ==' markers in prose becoming headings). - docs/content-model.md: the content management model (what is identified, versions and who did what (#51), tags and tag search (#52), storage). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
a9626b4d9e
commit
8d82527765
@ -39,7 +39,8 @@ test('prose: chapter heading, paragraphs, footnote', () => {
|
|||||||
assert.equal(prose.meta['مصنف'], 'سید احمد خان');
|
assert.equal(prose.meta['مصنف'], 'سید احمد خان');
|
||||||
assert.deepEqual(prose.blocks.map((b) => b.type), ['heading', 'para', 'para', 'para']);
|
assert.deepEqual(prose.blocks.map((b) => b.type), ['heading', 'para', 'para', 'para']);
|
||||||
assert.equal((prose.blocks[1] as any).line.notes.length, 1);
|
assert.equal((prose.blocks[1] as any).line.notes.length, 1);
|
||||||
assert.deepEqual(toVerses(prose).map((v) => v.Position), ['Paragraph', 'Paragraph', 'Paragraph']);
|
assert.deepEqual(toVerses(prose).map((v) => v.Position), ['Heading', 'Paragraph', 'Paragraph', 'Paragraph']);
|
||||||
|
assert.equal(toVerses(parse('== باب ==\n\n=== فصل ===\n\nمتن')).map((v) => v.Level ?? 0).join(), '2,3,0', 'chapter and sub-heading levels');
|
||||||
});
|
});
|
||||||
|
|
||||||
test('stanzas and single lines', () => {
|
test('stanzas and single lines', () => {
|
||||||
|
|||||||
@ -17,7 +17,7 @@
|
|||||||
|
|
||||||
export type Inline = { text: string; words: { shown: string; lemma: string }[]; notes: string[]; variants: { shown: string; others: string[]; source?: string }[] };
|
export type Inline = { text: string; words: { shown: string; lemma: string }[]; notes: string[]; variants: { shown: string; others: string[]; source?: string }[] };
|
||||||
export type Block =
|
export type Block =
|
||||||
| { type: 'heading'; text: string }
|
| { type: 'heading'; text: string; level: number } // == chapter == is level 2, === sub-heading === level 3, …
|
||||||
| { type: 'para'; line: Inline }
|
| { type: 'para'; line: Inline }
|
||||||
| { type: 'couplet' | 'line' | 'stanza'; lines: Inline[]; label?: string };
|
| { type: 'couplet' | 'line' | 'stanza'; lines: Inline[]; label?: string };
|
||||||
export type Doc = { meta: Record<string, string>; blocks: Block[] };
|
export type Doc = { meta: Record<string, string>; blocks: Block[] };
|
||||||
@ -76,8 +76,8 @@ export function parse(src: string): Doc {
|
|||||||
if (!lines.length) continue;
|
if (!lines.length) continue;
|
||||||
if (!verse) {
|
if (!verse) {
|
||||||
for (const l of lines) {
|
for (const l of lines) {
|
||||||
const h = l.match(/^=+\s*(.+?)\s*=+$/);
|
const h = l.match(/^(=+)\s*(.+?)\s*=+$/);
|
||||||
blocks.push(h ? { type: 'heading', text: h[1] } : { type: 'para', line: inline(l) });
|
blocks.push(h ? { type: 'heading', text: h[2], level: h[1].length } : { type: 'para', line: inline(l) });
|
||||||
}
|
}
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@ -98,10 +98,11 @@ export const words = (text: string) =>
|
|||||||
|
|
||||||
// → the site's verse JSON (the ganjoor-data layout divan-data already exports)
|
// → the site's verse JSON (the ganjoor-data layout divan-data already exports)
|
||||||
export function toVerses(doc: Doc) {
|
export function toVerses(doc: Doc) {
|
||||||
const out: { VOrder: number; Position: string; Text: string; CoupletIndex: number }[] = [];
|
const out: { VOrder: number; Position: string; Text: string; CoupletIndex: number; Level?: number }[] = [];
|
||||||
let couplet = 0;
|
let couplet = 0;
|
||||||
for (const b of doc.blocks) {
|
for (const b of doc.blocks) {
|
||||||
if (b.type === 'heading') continue;
|
// headings inside a work (chapter, sub-heading) are their own kind of verse entry
|
||||||
|
if (b.type === 'heading') { out.push({ VOrder: 0, Position: 'Heading', Text: b.text, CoupletIndex: couplet++, Level: b.level }); continue; }
|
||||||
if (b.type === 'para') out.push({ VOrder: 0, Position: 'Paragraph', Text: b.line.text, CoupletIndex: couplet++ });
|
if (b.type === 'para') out.push({ VOrder: 0, Position: 'Paragraph', Text: b.line.text, CoupletIndex: couplet++ });
|
||||||
else if (b.type === 'couplet') { b.lines.forEach((l, i) => out.push({ VOrder: 0, Position: i ? 'Left' : 'Right', Text: l.text, CoupletIndex: couplet })); couplet++; }
|
else if (b.type === 'couplet') { b.lines.forEach((l, i) => out.push({ VOrder: 0, Position: i ? 'Left' : 'Right', Text: l.text, CoupletIndex: couplet })); couplet++; }
|
||||||
else for (const l of b.lines) out.push({ VOrder: 0, Position: 'Single', Text: l.text, CoupletIndex: couplet++ });
|
else for (const l of b.lines) out.push({ VOrder: 0, Position: 'Single', Text: l.text, CoupletIndex: couplet++ });
|
||||||
@ -123,7 +124,7 @@ export function toTEI(doc: Doc) {
|
|||||||
const m = doc.meta, body: string[] = [];
|
const m = doc.meta, body: string[] = [];
|
||||||
let n = 0;
|
let n = 0;
|
||||||
for (const b of doc.blocks) {
|
for (const b of doc.blocks) {
|
||||||
if (b.type === 'heading') body.push(`<head>${x(b.text)}</head>`);
|
if (b.type === 'heading') body.push(`<head type="h${b.level}">${x(b.text)}</head>`);
|
||||||
else if (b.type === 'para') body.push(`<p>${teiLine(b.line)}</p>`);
|
else if (b.type === 'para') body.push(`<p>${teiLine(b.line)}</p>`);
|
||||||
else {
|
else {
|
||||||
const type = b.type === 'couplet' ? 'sher' : b.type === 'stanza' ? 'band' : 'misra';
|
const type = b.type === 'couplet' ? 'sher' : b.type === 'stanza' ? 'band' : 'misra';
|
||||||
|
|||||||
51
api/src/owned.ts
Normal file
51
api/src/owned.ts
Normal file
@ -0,0 +1,51 @@
|
|||||||
|
// Divan-owned content (#30): works edited in Divan are kept in divan-data's divan/ folder, so the daily
|
||||||
|
// Wikisource sync never overwrites them (export_divan.py prefers them). Each work is two files:
|
||||||
|
// divan/<work url>.dtx Divan text, the readable source (see divantext.ts, docs/content-model.md)
|
||||||
|
// divan/<work url>.json generated from it: Title, Verses (the site's verse layout), Edited (who, when)
|
||||||
|
// Publishing (#34) calls writeOwned and commits both; fromPoem gives the starting text for an edit.
|
||||||
|
// node src/owned.ts <divan-data dir> <work url> <file.dtx> [editor] write one by hand
|
||||||
|
import { mkdir, readFile, writeFile } from 'node:fs/promises';
|
||||||
|
import { dirname, join } from 'node:path';
|
||||||
|
import { parse, toVerses } from './divantext.ts';
|
||||||
|
|
||||||
|
type Verse = { Position: string; Text: string; CoupletIndex: number };
|
||||||
|
|
||||||
|
// a work (ganjoor-data poem JSON) as Divan text: verse in <poem> (blank line between couplets, stanzas and
|
||||||
|
// lines), prose as paragraphs
|
||||||
|
export function fromPoem(poem: { Title: string; Verses: Verse[]; SourceUrl?: string }, meta: Record<string, string> = {}) {
|
||||||
|
const head = Object.entries({ عنوان: poem.Title, ...meta, ...(poem.SourceUrl && { ماخذ: 'ویکی ماخذ', ماخذ_ربط: poem.SourceUrl }) })
|
||||||
|
.map(([k, v]) => `| ${k} = ${v}`).join('\n');
|
||||||
|
const units = new Map<number, Verse[]>();
|
||||||
|
for (const v of poem.Verses) (units.get(v.CoupletIndex) ?? units.set(v.CoupletIndex, []).get(v.CoupletIndex)!).push(v);
|
||||||
|
const parts: string[] = [];
|
||||||
|
let verse: string[] = [];
|
||||||
|
const flush = () => { if (verse.length) parts.push(`<poem>\n${verse.join('\n\n')}\n</poem>`); verse = []; };
|
||||||
|
for (const u of units.values()) {
|
||||||
|
if (u[0].Position === 'Paragraph') { flush(); parts.push(u.map((v) => v.Text).join('\n')); }
|
||||||
|
else if (u[0].Position === 'Heading') { flush(); parts.push(`${'='.repeat((u[0] as any).Level ?? 2)} ${u[0].Text} ${'='.repeat((u[0] as any).Level ?? 2)}`); }
|
||||||
|
else verse.push(u.map((v) => v.Text).join('\n'));
|
||||||
|
}
|
||||||
|
flush();
|
||||||
|
return `{{دیوان\n${head}\n}}\n${parts.join('\n\n')}\n`;
|
||||||
|
}
|
||||||
|
|
||||||
|
// write a Divan-owned work (the .dtx and the generated .json); returns the paths written
|
||||||
|
export async function writeOwned(dataDir: string, url: string, dtx: string, edited: { by: string; at?: string }) {
|
||||||
|
const doc = parse(dtx);
|
||||||
|
const verses = toVerses(doc);
|
||||||
|
if (!verses.length) throw new Error('the text has no verses or paragraphs');
|
||||||
|
const base = join(dataDir, 'divan', url.replace(/^\/+|\/+$/g, ''));
|
||||||
|
if (!/^[\w/-]+$/.test(url) || url.includes('..')) throw new Error(`not a work url: ${url}`);
|
||||||
|
await mkdir(dirname(base), { recursive: true });
|
||||||
|
const json = { FullUrl: '/' + url.replace(/^\/+/, ''), Title: doc.meta['عنوان'] ?? '', Verses: verses.map(({ VOrder, ...v }) => ({ VOrder, ...v, SectionIndex1: 0 })),
|
||||||
|
Edited: { by: edited.by, at: edited.at ?? new Date().toISOString() } };
|
||||||
|
await writeFile(base + '.dtx', dtx.endsWith('\n') ? dtx : dtx + '\n');
|
||||||
|
await writeFile(base + '.json', JSON.stringify(json, null, 1) + '\n');
|
||||||
|
return [base + '.dtx', base + '.json'];
|
||||||
|
}
|
||||||
|
|
||||||
|
if (import.meta.url === `file://${process.argv[1]}`) {
|
||||||
|
const [dir, url, file, by = 'divan'] = process.argv.slice(2);
|
||||||
|
if (!dir || !url || !file) { console.error('usage: node src/owned.ts <divan-data dir> <work url> <file.dtx> [editor]'); process.exit(1); }
|
||||||
|
console.log((await writeOwned(dir, url, await readFile(file, 'utf8'), { by })).join('\n'));
|
||||||
|
}
|
||||||
@ -115,6 +115,61 @@ above.
|
|||||||
8. **Later**: Wikibase as linked-data records for poets and works, if that becomes useful; OpenITI-style
|
8. **Later**: Wikibase as linked-data records for poets and works, if that becomes useful; OpenITI-style
|
||||||
stable identifiers for works and versions.
|
stable identifiers for works and versions.
|
||||||
|
|
||||||
|
## Content management model
|
||||||
|
|
||||||
|
Owner direction (October 2026): this is **content management**, not only text editing. Every piece of content is
|
||||||
|
identified for what it is, every change is a version with who did what, and content carries searchable tags.
|
||||||
|
|
||||||
|
### What is identified
|
||||||
|
|
||||||
|
| Entity | What it is | Its text or fields |
|
||||||
|
|---|---|---|
|
||||||
|
| شاعر / ادیب (poet, writer) | A person | Name, pen name, years, places; links to their books |
|
||||||
|
| تعارف (intro) | A poet's or a book's introduction | Divan text (paragraphs, headings) |
|
||||||
|
| کتاب (book) | A published collection | Title, year, publisher/edition, order of chapters; intro |
|
||||||
|
| باب / ذیلی باب (chapter, sub-chapter) | A section of a book, nested | Title, order of works |
|
||||||
|
| کلام (work) | A ghazal, nazm, rubai, prose chapter, … | Data points (genre, radif, metre, sources) and its text in Divan text |
|
||||||
|
| Inside a work's text | Identified by Divan text: chapter **heading** (level 2), **sub-heading** (level 3, 4, …), **شعر** (couplet) of two **مصرع** (lines), **بند** (stanza), single line, **paragraph**, **footnote**, **variant reading**, **word** (with its dictionary lemma) | Divan text |
|
||||||
|
| لغت (dictionary entry) | A word's meanings, readings, pronunciation | Fields |
|
||||||
|
| ٹیگ (tag) | A typed label | Type and name |
|
||||||
|
|
||||||
|
Every entity has a stable id. Elements inside a work (couplets, paragraphs, headings) are addressed by their
|
||||||
|
position, as the site already does with `#c3` links and bookmarks. The prototype shows the whole library can be
|
||||||
|
written this way: all 11,087 works convert to Divan text and back unchanged (the only clean-ups are repeated spaces
|
||||||
|
and raw `== … ==` markers in prose becoming real chapter headings).
|
||||||
|
|
||||||
|
### Versions: who did what (#51)
|
||||||
|
|
||||||
|
- **Revisions**: every change to any entity is a revision. It records the entity, the version number, the full
|
||||||
|
content (Divan text or fields), the author, the time, an edit summary and its status. Statuses are draft,
|
||||||
|
submitted, L1-approved, published, returned and rejected; reviewers and publishers are recorded too.
|
||||||
|
- **History** page per entity: each version with who, when, summary and status. Any two versions can be compared
|
||||||
|
with a **diff** (by line for Divan text, by field for metadata). **Revert** creates a new revision through the
|
||||||
|
pipeline.
|
||||||
|
- **Moderation log** across the site: who did what, filtered by person, entity, date or action.
|
||||||
|
- **Pipeline**: the approval pipeline (#31) moves revisions through their statuses. Publishing writes the published
|
||||||
|
version to divan-data and commits it (#34), so the git history mirrors the published revisions.
|
||||||
|
|
||||||
|
### Tags (#52)
|
||||||
|
|
||||||
|
- **Types**: tags are typed: موضوع (theme), صنف (genre), بحر (metre), شخصیت (person mentioned), مقام (place),
|
||||||
|
دور (period), and free tags.
|
||||||
|
- **Where**: they attach to any entity, down to a couplet or a phrase.
|
||||||
|
- **Finding tagged content**:
|
||||||
|
- every tag has a page listing everything tagged with it;
|
||||||
|
- tags are a search filter and facet, like authors;
|
||||||
|
- `tag:تصوف` can be typed in the search box.
|
||||||
|
- **History and permissions**: adding or removing a tag is a revision like any other change, so its history shows
|
||||||
|
who tagged what and when. The permission model gets a `tags` content type.
|
||||||
|
|
||||||
|
### Storage
|
||||||
|
|
||||||
|
- **PostgreSQL** holds the entities (`poets`, `categories`, `poems`, `verses` today), plus `revisions`, `tags` and
|
||||||
|
`entity_tags`.
|
||||||
|
- **divan-data** holds the published versions: works as `.dtx` with generated `.json` in `divan/` (#30, done in
|
||||||
|
this step). Intros, books and chapters follow the same pattern as they become editable, and tags go in the
|
||||||
|
Divan text header (`| ٹیگ = عشق، تصوف`) and a tags list.
|
||||||
|
|
||||||
## Prototype results
|
## Prototype results
|
||||||
|
|
||||||
`api/src/divantext.ts` parses Divan text into blocks, converts them to the site's verse JSON, and exports TEI.
|
`api/src/divantext.ts` parses Divan text into blocks, converts them to the site's verse JSON, and exports TEI.
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user