diff --git a/RMuseum/Models/Ganjoor/PublicExport/PublicExportDtos.cs b/RMuseum/Models/Ganjoor/PublicExport/PublicExportDtos.cs index 58ef5965..eb14f95b 100644 --- a/RMuseum/Models/Ganjoor/PublicExport/PublicExportDtos.cs +++ b/RMuseum/Models/Ganjoor/PublicExport/PublicExportDtos.cs @@ -24,9 +24,31 @@ namespace RMuseum.Models.Ganjoor.PublicExport public int PoemsCount { get; set; } + /// + /// number of ids grouped into each id-index shard file (see UrlTemplates.CatIdIndexShard / + /// PoemIdIndexShard). A consumer resolving id X fetches shard file "{X / IdIndexShardSize}.json". + /// + public int IdIndexShardSize { get; set; } + + /// + /// URL patterns for every file kind in this export, so an app can treat this repo as an + /// API without having to read the export source code. {placeholders} are literal. + /// + public PublicExportUrlTemplatesDto UrlTemplates { get; set; } = new PublicExportUrlTemplatesDto(); + public List Poets { get; set; } = new List(); } + public class PublicExportUrlTemplatesDto + { + public string Poet { get; set; } = "poets/{poetSlug}/poet.json"; + public string Category { get; set; } = "poets/{poetSlug}/{catPath}/_cat.json"; + public string Poem { get; set; } = "poets/{poetSlug}/{catPath}/{poemSlug}.json"; + public string PoetIdIndex { get; set; } = "index/poets-by-id.json"; + public string CatIdIndexShard { get; set; } = "index/cats-by-id/{bucket}.json"; + public string PoemIdIndexShard { get; set; } = "index/poems-by-id/{bucket}.json"; + } + public class PublicExportManifestPoetEntryDto { public int Id { get; set; } diff --git a/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-PublicDataExport.cs b/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-PublicDataExport.cs index 8569ad82..0508b22e 100644 --- a/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-PublicDataExport.cs +++ b/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-PublicDataExport.cs @@ -18,6 +18,14 @@ namespace RMuseum.Services.Implementation /// public partial class GanjoorService : IGanjoorService { + /// + /// how many ids are grouped into each id-index shard file — kept small enough that a + /// shard stays a cheap single fetch, large enough that the id-space doesn't produce an + /// unreasonable number of tiny files. 2000 ids/shard means ~500 shard files for Ganjoor's + /// current poem count. + /// + private const int IdIndexShardSize = 2000; + /// /// start exporting all published Ganjoor data (poets/categories/poems/verses) to a /// git-tracked JSON tree and pushing it to the configured remote. User-linked tables @@ -61,8 +69,13 @@ namespace RMuseum.Services.Implementation var manifest = new PublicExportManifestDto { GeneratedAtUtc = DateTime.UtcNow.ToString("O"), + IdIndexShardSize = IdIndexShardSize, }; + var poetIdIndex = new Dictionary(); + var catIdIndex = new Dictionary(); + var poemIdIndex = new Dictionary(); + int poetIndex = 0; foreach (var poet in poets) { @@ -76,8 +89,9 @@ namespace RMuseum.Services.Implementation continue; await ExportPoetToJson(context, repoRoot, poet, catPoet); + poetIdIndex[poet.Id] = catPoet.FullUrl; - int poemCount = await ExportCatTreeToJson(context, repoRoot, catPoet); + int poemCount = await ExportCatTreeToJson(context, repoRoot, catPoet, catIdIndex, poemIdIndex); manifest.PoemsCount += poemCount; manifest.Poets.Add(new PublicExportManifestPoetEntryDto @@ -90,7 +104,13 @@ namespace RMuseum.Services.Implementation manifest.PoetsCount = manifest.Poets.Count; + await jobProgressServiceEF.UpdateJob(job.Id, 98, "Writing id indexes"); + await IdIndexWriter.WriteFlatIndexAsync(repoRoot, "index/poets-by-id.json", poetIdIndex); + await IdIndexWriter.WriteShardedIndexAsync(repoRoot, "cats", catIdIndex, IdIndexShardSize); + await IdIndexWriter.WriteShardedIndexAsync(repoRoot, "poems", poemIdIndex, IdIndexShardSize); + await DeterministicJsonWriter.WriteIfChangedAsync(Path.Combine(repoRoot, "manifest.json"), manifest); + await TextFileWriter.WriteIfChangedAsync(Path.Combine(repoRoot, "API.md"), BuildApiMarkdown(manifest)); await jobProgressServiceEF.UpdateJob(job.Id, 99, "Committing and pushing"); int changed = publisher.CommitAndPush($"data: export {manifest.PoetsCount} poets / {manifest.PoemsCount} poems — {DateTime.UtcNow:yyyy-MM-dd}"); @@ -185,7 +205,8 @@ namespace RMuseum.Services.Implementation /// under it, then recurses into published child categories. Returns the number of poems written /// in this subtree (for manifest counts). /// - private async Task ExportCatTreeToJson(RMuseumDbContext context, string repoRoot, GanjoorCat cat) + private async Task ExportCatTreeToJson(RMuseumDbContext context, string repoRoot, GanjoorCat cat, + Dictionary catIdIndex, Dictionary poemIdIndex) { var childCats = await context.GanjoorCategories.AsNoTracking() .Where(c => c.ParentId == cat.Id && c.Published) @@ -197,6 +218,8 @@ namespace RMuseum.Services.Implementation .OrderBy(p => p.Id) .ToListAsync(); + catIdIndex[cat.Id] = cat.FullUrl; + var catDto = new CatPublicDto { Id = cat.Id, @@ -219,11 +242,12 @@ namespace RMuseum.Services.Implementation foreach (var poem in poems) { await ExportPoemToJson(context, repoRoot, poem); + poemIdIndex[poem.Id] = poem.FullUrl; } foreach (var childCat in childCats) { - poemCount += await ExportCatTreeToJson(context, repoRoot, childCat); + poemCount += await ExportCatTreeToJson(context, repoRoot, childCat, catIdIndex, poemIdIndex); } return poemCount; @@ -294,5 +318,68 @@ namespace RMuseum.Services.Implementation url = url.Replace('/', Path.DirectorySeparatorChar); return url.TrimStart(Path.DirectorySeparatorChar); } + + /// + /// generates the repo-root API.md every run, so the docs can never drift out of sync with + /// UrlTemplates/IdIndexShardSize in manifest.json. Not hand-edited — if you want to add + /// prose, add it here, not in the generated file. + /// + private static string BuildApiMarkdown(PublicExportManifestDto manifest) + { + var t = manifest.UrlTemplates; + return +$@"# Ganjoor public data — static API + +This repository is served as a static API through jsDelivr's GitHub CDN. There is no server — +every ""endpoint"" below is a file in this repo, fetched over plain HTTPS with CORS enabled. + +Base URL (tracks the latest commit on `main`): + + https://cdn.jsdelivr.net/gh/ganjoor/ganjoor-data@main/ + +Note: since this currently tracks `@main` rather than tagged releases, jsDelivr's edge cache +(up to ~7 days) means a fetch can lag behind the newest commit. If you need a frozen snapshot, +pin to an exact commit instead of `@main`: + + https://cdn.jsdelivr.net/gh/ganjoor/ganjoor-data@/ + +## Discovery + +`GET manifest.json` — schema version, generation timestamp, poet/poem counts, the list of poets +with their paths, and the URL templates below (`manifest.json` is the source of truth for this +file; if they ever disagree, trust `manifest.json`). + +## Content + +- `GET {t.Poet}` — poet biography +- `GET {t.Category}` — a category/collection: title, description, ordered child categories and poems +- `GET {t.Poem}` — a poem: metre, rhyme, sections, verses + +`{{poetSlug}}`/`{{catPath}}`/`{{poemSlug}}` are exactly the path segments of the poem's Ganjoor +URL, e.g. the poem at ganjoor.net/hafez/ghazal/sh1 is at `poets/hafez/ghazal/sh1.json`. + +## Resolving a numeric id + +If you have a poet/category/poem id (not a slug), resolve it via the id index instead of +guessing a path: + +- `GET {t.PoetIdIndex}` — small enough to fetch whole: `{{ ""1"": ""/hafez"", ... }}` +- `GET {t.CatIdIndexShard}` / `GET {t.PoemIdIndexShard}` — bucketed. The shard file for id `X` is + `bucket = X / {manifest.IdIndexShardSize}` (integer division), so e.g. poem id 4321 with the + current shard size of {manifest.IdIndexShardSize} lives in shard `{4321 / manifest.IdIndexShardSize}`, + i.e. `index/poems-by-id/{4321 / manifest.IdIndexShardSize}.json`. + Each shard maps ids in its range to a `FullUrl` you then fetch with the `{t.Poem}` pattern above. + +## Not included + +No comments, bookmarks, reading history, edit/correction history, or any other user-account-linked +data — see the repo this data set is exported from for details. Only `Published` poets/categories/ +poems are included. + +## Search + +Not available as a static endpoint in this data set yet. +"; + } } } diff --git a/RMuseum/Utils/PublicDataExport/IdIndexWriter.cs b/RMuseum/Utils/PublicDataExport/IdIndexWriter.cs new file mode 100644 index 00000000..9eae470f --- /dev/null +++ b/RMuseum/Utils/PublicDataExport/IdIndexWriter.cs @@ -0,0 +1,45 @@ +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Threading.Tasks; + +namespace RMuseum.Utils.PublicDataExport +{ + /// + /// A bare numeric id (poem id, category id, ...) is meaningless to a static file tree unless + /// something maps it to a path. Writing one giant id->path file doesn't scale to Ganjoor's + /// poem count, so ids are bucketed by id / shardSize into small shard files a client can + /// compute the name of directly — no lookup-before-the-lookup needed. + /// + public static class IdIndexWriter + { + /// + /// Writes every id in to poets-by-id.json (or configured file + /// name) with no sharding — for tables small enough that one file is fine (e.g. poets: a + /// few hundred rows). + /// + public static async Task WriteFlatIndexAsync(string repoRoot, string relativeFilePath, Dictionary idToPath) + { + var ordered = idToPath.OrderBy(kv => kv.Key).ToDictionary(kv => kv.Key.ToString(), kv => kv.Value); + await DeterministicJsonWriter.WriteIfChangedAsync(Path.Combine(repoRoot, relativeFilePath), ordered); + } + + /// + /// Writes as bucketed shard files under + /// /index/{category}-by-id/{bucket}.json, where + /// bucket = id / shardSize. Only buckets that actually contain ids get a file — an empty + /// bucket produces no request-able file, which is fine since a client only ever asks for + /// the bucket of an id it already has. + /// + public static async Task WriteShardedIndexAsync(string repoRoot, string category, Dictionary idToPath, int shardSize) + { + var byBucket = idToPath.GroupBy(kv => kv.Key / shardSize); + foreach (var bucket in byBucket) + { + var ordered = bucket.OrderBy(kv => kv.Key).ToDictionary(kv => kv.Key.ToString(), kv => kv.Value); + string path = Path.Combine(repoRoot, "index", $"{category}-by-id", $"{bucket.Key}.json"); + await DeterministicJsonWriter.WriteIfChangedAsync(path, ordered); + } + } + } +} diff --git a/RMuseum/Utils/PublicDataExport/TextFileWriter.cs b/RMuseum/Utils/PublicDataExport/TextFileWriter.cs new file mode 100644 index 00000000..dcb14b5a --- /dev/null +++ b/RMuseum/Utils/PublicDataExport/TextFileWriter.cs @@ -0,0 +1,44 @@ +using System.IO; +using System.Text; +using System.Threading.Tasks; + +namespace RMuseum.Utils.PublicDataExport +{ + /// + /// Same no-op-if-unchanged, LF/UTF-8-no-BOM behavior as , + /// for plain-text files (currently just the generated API.md) rather than JSON. + /// + public static class TextFileWriter + { + private static readonly UTF8Encoding _utf8NoBom = new UTF8Encoding(encoderShouldEmitUTF8Identifier: false); + + public static async Task WriteIfChangedAsync(string path, string content) + { + content = content.Replace("\r\n", "\n").TrimEnd('\n') + "\n"; + byte[] newBytes = _utf8NoBom.GetBytes(content); + + string dir = Path.GetDirectoryName(path); + if (!string.IsNullOrEmpty(dir) && !Directory.Exists(dir)) + { + Directory.CreateDirectory(dir); + } + + if (File.Exists(path)) + { + byte[] existingBytes = await File.ReadAllBytesAsync(path); + if (existingBytes.Length == newBytes.Length) + { + bool same = true; + for (int i = 0; i < existingBytes.Length; i++) + { + if (existingBytes[i] != newBytes[i]) { same = false; break; } + } + if (same) return false; + } + } + + await File.WriteAllBytesAsync(path, newBytes); + return true; + } + } +}