#498 export public data

This commit is contained in:
Hamid Reza Mohammadi 2026-08-14 13:49:57 +03:30
parent 363e1f5b27
commit 6768ee4c52
4 changed files with 201 additions and 3 deletions

View File

@ -24,9 +24,31 @@ namespace RMuseum.Models.Ganjoor.PublicExport
public int PoemsCount { get; set; }
/// <summary>
/// number of ids grouped into each id-index shard file (see UrlTemplates.CatIdIndexShard /
/// PoemIdIndexShard). A consumer resolving id X fetches shard file "{X / IdIndexShardSize}.json".
/// </summary>
public int IdIndexShardSize { get; set; }
/// <summary>
/// URL patterns for every file kind in this export, so an app can treat this repo as an
/// API without having to read the export source code. {placeholders} are literal.
/// </summary>
public PublicExportUrlTemplatesDto UrlTemplates { get; set; } = new PublicExportUrlTemplatesDto();
public List<PublicExportManifestPoetEntryDto> Poets { get; set; } = new List<PublicExportManifestPoetEntryDto>();
}
public class PublicExportUrlTemplatesDto
{
public string Poet { get; set; } = "poets/{poetSlug}/poet.json";
public string Category { get; set; } = "poets/{poetSlug}/{catPath}/_cat.json";
public string Poem { get; set; } = "poets/{poetSlug}/{catPath}/{poemSlug}.json";
public string PoetIdIndex { get; set; } = "index/poets-by-id.json";
public string CatIdIndexShard { get; set; } = "index/cats-by-id/{bucket}.json";
public string PoemIdIndexShard { get; set; } = "index/poems-by-id/{bucket}.json";
}
public class PublicExportManifestPoetEntryDto
{
public int Id { get; set; }

View File

@ -18,6 +18,14 @@ namespace RMuseum.Services.Implementation
/// </summary>
public partial class GanjoorService : IGanjoorService
{
/// <summary>
/// how many ids are grouped into each id-index shard file — kept small enough that a
/// shard stays a cheap single fetch, large enough that the id-space doesn't produce an
/// unreasonable number of tiny files. 2000 ids/shard means ~500 shard files for Ganjoor's
/// current poem count.
/// </summary>
private const int IdIndexShardSize = 2000;
/// <summary>
/// start exporting all published Ganjoor data (poets/categories/poems/verses) to a
/// git-tracked JSON tree and pushing it to the configured remote. User-linked tables
@ -61,8 +69,13 @@ namespace RMuseum.Services.Implementation
var manifest = new PublicExportManifestDto
{
GeneratedAtUtc = DateTime.UtcNow.ToString("O"),
IdIndexShardSize = IdIndexShardSize,
};
var poetIdIndex = new Dictionary<int, string>();
var catIdIndex = new Dictionary<int, string>();
var poemIdIndex = new Dictionary<int, string>();
int poetIndex = 0;
foreach (var poet in poets)
{
@ -76,8 +89,9 @@ namespace RMuseum.Services.Implementation
continue;
await ExportPoetToJson(context, repoRoot, poet, catPoet);
poetIdIndex[poet.Id] = catPoet.FullUrl;
int poemCount = await ExportCatTreeToJson(context, repoRoot, catPoet);
int poemCount = await ExportCatTreeToJson(context, repoRoot, catPoet, catIdIndex, poemIdIndex);
manifest.PoemsCount += poemCount;
manifest.Poets.Add(new PublicExportManifestPoetEntryDto
@ -90,7 +104,13 @@ namespace RMuseum.Services.Implementation
manifest.PoetsCount = manifest.Poets.Count;
await jobProgressServiceEF.UpdateJob(job.Id, 98, "Writing id indexes");
await IdIndexWriter.WriteFlatIndexAsync(repoRoot, "index/poets-by-id.json", poetIdIndex);
await IdIndexWriter.WriteShardedIndexAsync(repoRoot, "cats", catIdIndex, IdIndexShardSize);
await IdIndexWriter.WriteShardedIndexAsync(repoRoot, "poems", poemIdIndex, IdIndexShardSize);
await DeterministicJsonWriter.WriteIfChangedAsync(Path.Combine(repoRoot, "manifest.json"), manifest);
await TextFileWriter.WriteIfChangedAsync(Path.Combine(repoRoot, "API.md"), BuildApiMarkdown(manifest));
await jobProgressServiceEF.UpdateJob(job.Id, 99, "Committing and pushing");
int changed = publisher.CommitAndPush($"data: export {manifest.PoetsCount} poets / {manifest.PoemsCount} poems — {DateTime.UtcNow:yyyy-MM-dd}");
@ -185,7 +205,8 @@ namespace RMuseum.Services.Implementation
/// under it, then recurses into published child categories. Returns the number of poems written
/// in this subtree (for manifest counts).
/// </summary>
private async Task<int> ExportCatTreeToJson(RMuseumDbContext context, string repoRoot, GanjoorCat cat)
private async Task<int> ExportCatTreeToJson(RMuseumDbContext context, string repoRoot, GanjoorCat cat,
Dictionary<int, string> catIdIndex, Dictionary<int, string> poemIdIndex)
{
var childCats = await context.GanjoorCategories.AsNoTracking()
.Where(c => c.ParentId == cat.Id && c.Published)
@ -197,6 +218,8 @@ namespace RMuseum.Services.Implementation
.OrderBy(p => p.Id)
.ToListAsync();
catIdIndex[cat.Id] = cat.FullUrl;
var catDto = new CatPublicDto
{
Id = cat.Id,
@ -219,11 +242,12 @@ namespace RMuseum.Services.Implementation
foreach (var poem in poems)
{
await ExportPoemToJson(context, repoRoot, poem);
poemIdIndex[poem.Id] = poem.FullUrl;
}
foreach (var childCat in childCats)
{
poemCount += await ExportCatTreeToJson(context, repoRoot, childCat);
poemCount += await ExportCatTreeToJson(context, repoRoot, childCat, catIdIndex, poemIdIndex);
}
return poemCount;
@ -294,5 +318,68 @@ namespace RMuseum.Services.Implementation
url = url.Replace('/', Path.DirectorySeparatorChar);
return url.TrimStart(Path.DirectorySeparatorChar);
}
/// <summary>
/// generates the repo-root API.md every run, so the docs can never drift out of sync with
/// UrlTemplates/IdIndexShardSize in manifest.json. Not hand-edited — if you want to add
/// prose, add it here, not in the generated file.
/// </summary>
private static string BuildApiMarkdown(PublicExportManifestDto manifest)
{
var t = manifest.UrlTemplates;
return
$@"# Ganjoor public data — static API
This repository is served as a static API through jsDelivr's GitHub CDN. There is no server —
every ""endpoint"" below is a file in this repo, fetched over plain HTTPS with CORS enabled.
Base URL (tracks the latest commit on `main`):
https://cdn.jsdelivr.net/gh/ganjoor/ganjoor-data@main/
Note: since this currently tracks `@main` rather than tagged releases, jsDelivr's edge cache
(up to ~7 days) means a fetch can lag behind the newest commit. If you need a frozen snapshot,
pin to an exact commit instead of `@main`:
https://cdn.jsdelivr.net/gh/ganjoor/ganjoor-data@<commit-sha>/
## Discovery
`GET manifest.json` — schema version, generation timestamp, poet/poem counts, the list of poets
with their paths, and the URL templates below (`manifest.json` is the source of truth for this
file; if they ever disagree, trust `manifest.json`).
## Content
- `GET {t.Poet}` — poet biography
- `GET {t.Category}` — a category/collection: title, description, ordered child categories and poems
- `GET {t.Poem}` — a poem: metre, rhyme, sections, verses
`{{poetSlug}}`/`{{catPath}}`/`{{poemSlug}}` are exactly the path segments of the poem's Ganjoor
URL, e.g. the poem at ganjoor.net/hafez/ghazal/sh1 is at `poets/hafez/ghazal/sh1.json`.
## Resolving a numeric id
If you have a poet/category/poem id (not a slug), resolve it via the id index instead of
guessing a path:
- `GET {t.PoetIdIndex}` — small enough to fetch whole: `{{ ""1"": ""/hafez"", ... }}`
- `GET {t.CatIdIndexShard}` / `GET {t.PoemIdIndexShard}` — bucketed. The shard file for id `X` is
`bucket = X / {manifest.IdIndexShardSize}` (integer division), so e.g. poem id 4321 with the
current shard size of {manifest.IdIndexShardSize} lives in shard `{4321 / manifest.IdIndexShardSize}`,
i.e. `index/poems-by-id/{4321 / manifest.IdIndexShardSize}.json`.
Each shard maps ids in its range to a `FullUrl` you then fetch with the `{t.Poem}` pattern above.
## Not included
No comments, bookmarks, reading history, edit/correction history, or any other user-account-linked
data — see the repo this data set is exported from for details. Only `Published` poets/categories/
poems are included.
## Search
Not available as a static endpoint in this data set yet.
";
}
}
}

View File

@ -0,0 +1,45 @@
using System.Collections.Generic;
using System.IO;
using System.Linq;
using System.Threading.Tasks;
namespace RMuseum.Utils.PublicDataExport
{
/// <summary>
/// A bare numeric id (poem id, category id, ...) is meaningless to a static file tree unless
/// something maps it to a path. Writing one giant id-&gt;path file doesn't scale to Ganjoor's
/// poem count, so ids are bucketed by <c>id / shardSize</c> into small shard files a client can
/// compute the name of directly — no lookup-before-the-lookup needed.
/// </summary>
public static class IdIndexWriter
{
/// <summary>
/// Writes every id in <paramref name="idToPath"/> to poets-by-id.json (or configured file
/// name) with no sharding — for tables small enough that one file is fine (e.g. poets: a
/// few hundred rows).
/// </summary>
public static async Task WriteFlatIndexAsync(string repoRoot, string relativeFilePath, Dictionary<int, string> idToPath)
{
var ordered = idToPath.OrderBy(kv => kv.Key).ToDictionary(kv => kv.Key.ToString(), kv => kv.Value);
await DeterministicJsonWriter.WriteIfChangedAsync(Path.Combine(repoRoot, relativeFilePath), ordered);
}
/// <summary>
/// Writes <paramref name="idToPath"/> as bucketed shard files under
/// <paramref name="repoRoot"/>/index/{category}-by-id/{bucket}.json, where
/// bucket = id / shardSize. Only buckets that actually contain ids get a file — an empty
/// bucket produces no request-able file, which is fine since a client only ever asks for
/// the bucket of an id it already has.
/// </summary>
public static async Task WriteShardedIndexAsync(string repoRoot, string category, Dictionary<int, string> idToPath, int shardSize)
{
var byBucket = idToPath.GroupBy(kv => kv.Key / shardSize);
foreach (var bucket in byBucket)
{
var ordered = bucket.OrderBy(kv => kv.Key).ToDictionary(kv => kv.Key.ToString(), kv => kv.Value);
string path = Path.Combine(repoRoot, "index", $"{category}-by-id", $"{bucket.Key}.json");
await DeterministicJsonWriter.WriteIfChangedAsync(path, ordered);
}
}
}
}

View File

@ -0,0 +1,44 @@
using System.IO;
using System.Text;
using System.Threading.Tasks;
namespace RMuseum.Utils.PublicDataExport
{
/// <summary>
/// Same no-op-if-unchanged, LF/UTF-8-no-BOM behavior as <see cref="DeterministicJsonWriter"/>,
/// for plain-text files (currently just the generated API.md) rather than JSON.
/// </summary>
public static class TextFileWriter
{
private static readonly UTF8Encoding _utf8NoBom = new UTF8Encoding(encoderShouldEmitUTF8Identifier: false);
public static async Task<bool> WriteIfChangedAsync(string path, string content)
{
content = content.Replace("\r\n", "\n").TrimEnd('\n') + "\n";
byte[] newBytes = _utf8NoBom.GetBytes(content);
string dir = Path.GetDirectoryName(path);
if (!string.IsNullOrEmpty(dir) && !Directory.Exists(dir))
{
Directory.CreateDirectory(dir);
}
if (File.Exists(path))
{
byte[] existingBytes = await File.ReadAllBytesAsync(path);
if (existingBytes.Length == newBytes.Length)
{
bool same = true;
for (int i = 0; i < existingBytes.Length; i++)
{
if (existingBytes[i] != newBytes[i]) { same = false; break; }
}
if (same) return false;
}
}
await File.WriteAllBytesAsync(path, newBytes);
return true;
}
}
}