#498 export public data
This commit is contained in:
parent
363e1f5b27
commit
6768ee4c52
@ -24,9 +24,31 @@ namespace RMuseum.Models.Ganjoor.PublicExport
|
||||
|
||||
public int PoemsCount { get; set; }
|
||||
|
||||
/// <summary>
|
||||
/// number of ids grouped into each id-index shard file (see UrlTemplates.CatIdIndexShard /
|
||||
/// PoemIdIndexShard). A consumer resolving id X fetches shard file "{X / IdIndexShardSize}.json".
|
||||
/// </summary>
|
||||
public int IdIndexShardSize { get; set; }
|
||||
|
||||
/// <summary>
|
||||
/// URL patterns for every file kind in this export, so an app can treat this repo as an
|
||||
/// API without having to read the export source code. {placeholders} are literal.
|
||||
/// </summary>
|
||||
public PublicExportUrlTemplatesDto UrlTemplates { get; set; } = new PublicExportUrlTemplatesDto();
|
||||
|
||||
public List<PublicExportManifestPoetEntryDto> Poets { get; set; } = new List<PublicExportManifestPoetEntryDto>();
|
||||
}
|
||||
|
||||
public class PublicExportUrlTemplatesDto
|
||||
{
|
||||
public string Poet { get; set; } = "poets/{poetSlug}/poet.json";
|
||||
public string Category { get; set; } = "poets/{poetSlug}/{catPath}/_cat.json";
|
||||
public string Poem { get; set; } = "poets/{poetSlug}/{catPath}/{poemSlug}.json";
|
||||
public string PoetIdIndex { get; set; } = "index/poets-by-id.json";
|
||||
public string CatIdIndexShard { get; set; } = "index/cats-by-id/{bucket}.json";
|
||||
public string PoemIdIndexShard { get; set; } = "index/poems-by-id/{bucket}.json";
|
||||
}
|
||||
|
||||
public class PublicExportManifestPoetEntryDto
|
||||
{
|
||||
public int Id { get; set; }
|
||||
|
||||
@ -18,6 +18,14 @@ namespace RMuseum.Services.Implementation
|
||||
/// </summary>
|
||||
public partial class GanjoorService : IGanjoorService
|
||||
{
|
||||
/// <summary>
|
||||
/// how many ids are grouped into each id-index shard file — kept small enough that a
|
||||
/// shard stays a cheap single fetch, large enough that the id-space doesn't produce an
|
||||
/// unreasonable number of tiny files. 2000 ids/shard means ~500 shard files for Ganjoor's
|
||||
/// current poem count.
|
||||
/// </summary>
|
||||
private const int IdIndexShardSize = 2000;
|
||||
|
||||
/// <summary>
|
||||
/// start exporting all published Ganjoor data (poets/categories/poems/verses) to a
|
||||
/// git-tracked JSON tree and pushing it to the configured remote. User-linked tables
|
||||
@ -61,8 +69,13 @@ namespace RMuseum.Services.Implementation
|
||||
var manifest = new PublicExportManifestDto
|
||||
{
|
||||
GeneratedAtUtc = DateTime.UtcNow.ToString("O"),
|
||||
IdIndexShardSize = IdIndexShardSize,
|
||||
};
|
||||
|
||||
var poetIdIndex = new Dictionary<int, string>();
|
||||
var catIdIndex = new Dictionary<int, string>();
|
||||
var poemIdIndex = new Dictionary<int, string>();
|
||||
|
||||
int poetIndex = 0;
|
||||
foreach (var poet in poets)
|
||||
{
|
||||
@ -76,8 +89,9 @@ namespace RMuseum.Services.Implementation
|
||||
continue;
|
||||
|
||||
await ExportPoetToJson(context, repoRoot, poet, catPoet);
|
||||
poetIdIndex[poet.Id] = catPoet.FullUrl;
|
||||
|
||||
int poemCount = await ExportCatTreeToJson(context, repoRoot, catPoet);
|
||||
int poemCount = await ExportCatTreeToJson(context, repoRoot, catPoet, catIdIndex, poemIdIndex);
|
||||
manifest.PoemsCount += poemCount;
|
||||
|
||||
manifest.Poets.Add(new PublicExportManifestPoetEntryDto
|
||||
@ -90,7 +104,13 @@ namespace RMuseum.Services.Implementation
|
||||
|
||||
manifest.PoetsCount = manifest.Poets.Count;
|
||||
|
||||
await jobProgressServiceEF.UpdateJob(job.Id, 98, "Writing id indexes");
|
||||
await IdIndexWriter.WriteFlatIndexAsync(repoRoot, "index/poets-by-id.json", poetIdIndex);
|
||||
await IdIndexWriter.WriteShardedIndexAsync(repoRoot, "cats", catIdIndex, IdIndexShardSize);
|
||||
await IdIndexWriter.WriteShardedIndexAsync(repoRoot, "poems", poemIdIndex, IdIndexShardSize);
|
||||
|
||||
await DeterministicJsonWriter.WriteIfChangedAsync(Path.Combine(repoRoot, "manifest.json"), manifest);
|
||||
await TextFileWriter.WriteIfChangedAsync(Path.Combine(repoRoot, "API.md"), BuildApiMarkdown(manifest));
|
||||
|
||||
await jobProgressServiceEF.UpdateJob(job.Id, 99, "Committing and pushing");
|
||||
int changed = publisher.CommitAndPush($"data: export {manifest.PoetsCount} poets / {manifest.PoemsCount} poems — {DateTime.UtcNow:yyyy-MM-dd}");
|
||||
@ -185,7 +205,8 @@ namespace RMuseum.Services.Implementation
|
||||
/// under it, then recurses into published child categories. Returns the number of poems written
|
||||
/// in this subtree (for manifest counts).
|
||||
/// </summary>
|
||||
private async Task<int> ExportCatTreeToJson(RMuseumDbContext context, string repoRoot, GanjoorCat cat)
|
||||
private async Task<int> ExportCatTreeToJson(RMuseumDbContext context, string repoRoot, GanjoorCat cat,
|
||||
Dictionary<int, string> catIdIndex, Dictionary<int, string> poemIdIndex)
|
||||
{
|
||||
var childCats = await context.GanjoorCategories.AsNoTracking()
|
||||
.Where(c => c.ParentId == cat.Id && c.Published)
|
||||
@ -197,6 +218,8 @@ namespace RMuseum.Services.Implementation
|
||||
.OrderBy(p => p.Id)
|
||||
.ToListAsync();
|
||||
|
||||
catIdIndex[cat.Id] = cat.FullUrl;
|
||||
|
||||
var catDto = new CatPublicDto
|
||||
{
|
||||
Id = cat.Id,
|
||||
@ -219,11 +242,12 @@ namespace RMuseum.Services.Implementation
|
||||
foreach (var poem in poems)
|
||||
{
|
||||
await ExportPoemToJson(context, repoRoot, poem);
|
||||
poemIdIndex[poem.Id] = poem.FullUrl;
|
||||
}
|
||||
|
||||
foreach (var childCat in childCats)
|
||||
{
|
||||
poemCount += await ExportCatTreeToJson(context, repoRoot, childCat);
|
||||
poemCount += await ExportCatTreeToJson(context, repoRoot, childCat, catIdIndex, poemIdIndex);
|
||||
}
|
||||
|
||||
return poemCount;
|
||||
@ -294,5 +318,68 @@ namespace RMuseum.Services.Implementation
|
||||
url = url.Replace('/', Path.DirectorySeparatorChar);
|
||||
return url.TrimStart(Path.DirectorySeparatorChar);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// generates the repo-root API.md every run, so the docs can never drift out of sync with
|
||||
/// UrlTemplates/IdIndexShardSize in manifest.json. Not hand-edited — if you want to add
|
||||
/// prose, add it here, not in the generated file.
|
||||
/// </summary>
|
||||
private static string BuildApiMarkdown(PublicExportManifestDto manifest)
|
||||
{
|
||||
var t = manifest.UrlTemplates;
|
||||
return
|
||||
$@"# Ganjoor public data — static API
|
||||
|
||||
This repository is served as a static API through jsDelivr's GitHub CDN. There is no server —
|
||||
every ""endpoint"" below is a file in this repo, fetched over plain HTTPS with CORS enabled.
|
||||
|
||||
Base URL (tracks the latest commit on `main`):
|
||||
|
||||
https://cdn.jsdelivr.net/gh/ganjoor/ganjoor-data@main/
|
||||
|
||||
Note: since this currently tracks `@main` rather than tagged releases, jsDelivr's edge cache
|
||||
(up to ~7 days) means a fetch can lag behind the newest commit. If you need a frozen snapshot,
|
||||
pin to an exact commit instead of `@main`:
|
||||
|
||||
https://cdn.jsdelivr.net/gh/ganjoor/ganjoor-data@<commit-sha>/
|
||||
|
||||
## Discovery
|
||||
|
||||
`GET manifest.json` — schema version, generation timestamp, poet/poem counts, the list of poets
|
||||
with their paths, and the URL templates below (`manifest.json` is the source of truth for this
|
||||
file; if they ever disagree, trust `manifest.json`).
|
||||
|
||||
## Content
|
||||
|
||||
- `GET {t.Poet}` — poet biography
|
||||
- `GET {t.Category}` — a category/collection: title, description, ordered child categories and poems
|
||||
- `GET {t.Poem}` — a poem: metre, rhyme, sections, verses
|
||||
|
||||
`{{poetSlug}}`/`{{catPath}}`/`{{poemSlug}}` are exactly the path segments of the poem's Ganjoor
|
||||
URL, e.g. the poem at ganjoor.net/hafez/ghazal/sh1 is at `poets/hafez/ghazal/sh1.json`.
|
||||
|
||||
## Resolving a numeric id
|
||||
|
||||
If you have a poet/category/poem id (not a slug), resolve it via the id index instead of
|
||||
guessing a path:
|
||||
|
||||
- `GET {t.PoetIdIndex}` — small enough to fetch whole: `{{ ""1"": ""/hafez"", ... }}`
|
||||
- `GET {t.CatIdIndexShard}` / `GET {t.PoemIdIndexShard}` — bucketed. The shard file for id `X` is
|
||||
`bucket = X / {manifest.IdIndexShardSize}` (integer division), so e.g. poem id 4321 with the
|
||||
current shard size of {manifest.IdIndexShardSize} lives in shard `{4321 / manifest.IdIndexShardSize}`,
|
||||
i.e. `index/poems-by-id/{4321 / manifest.IdIndexShardSize}.json`.
|
||||
Each shard maps ids in its range to a `FullUrl` you then fetch with the `{t.Poem}` pattern above.
|
||||
|
||||
## Not included
|
||||
|
||||
No comments, bookmarks, reading history, edit/correction history, or any other user-account-linked
|
||||
data — see the repo this data set is exported from for details. Only `Published` poets/categories/
|
||||
poems are included.
|
||||
|
||||
## Search
|
||||
|
||||
Not available as a static endpoint in this data set yet.
|
||||
";
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
45
RMuseum/Utils/PublicDataExport/IdIndexWriter.cs
Normal file
45
RMuseum/Utils/PublicDataExport/IdIndexWriter.cs
Normal file
@ -0,0 +1,45 @@
|
||||
using System.Collections.Generic;
|
||||
using System.IO;
|
||||
using System.Linq;
|
||||
using System.Threading.Tasks;
|
||||
|
||||
namespace RMuseum.Utils.PublicDataExport
|
||||
{
|
||||
/// <summary>
|
||||
/// A bare numeric id (poem id, category id, ...) is meaningless to a static file tree unless
|
||||
/// something maps it to a path. Writing one giant id->path file doesn't scale to Ganjoor's
|
||||
/// poem count, so ids are bucketed by <c>id / shardSize</c> into small shard files a client can
|
||||
/// compute the name of directly — no lookup-before-the-lookup needed.
|
||||
/// </summary>
|
||||
public static class IdIndexWriter
|
||||
{
|
||||
/// <summary>
|
||||
/// Writes every id in <paramref name="idToPath"/> to poets-by-id.json (or configured file
|
||||
/// name) with no sharding — for tables small enough that one file is fine (e.g. poets: a
|
||||
/// few hundred rows).
|
||||
/// </summary>
|
||||
public static async Task WriteFlatIndexAsync(string repoRoot, string relativeFilePath, Dictionary<int, string> idToPath)
|
||||
{
|
||||
var ordered = idToPath.OrderBy(kv => kv.Key).ToDictionary(kv => kv.Key.ToString(), kv => kv.Value);
|
||||
await DeterministicJsonWriter.WriteIfChangedAsync(Path.Combine(repoRoot, relativeFilePath), ordered);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Writes <paramref name="idToPath"/> as bucketed shard files under
|
||||
/// <paramref name="repoRoot"/>/index/{category}-by-id/{bucket}.json, where
|
||||
/// bucket = id / shardSize. Only buckets that actually contain ids get a file — an empty
|
||||
/// bucket produces no request-able file, which is fine since a client only ever asks for
|
||||
/// the bucket of an id it already has.
|
||||
/// </summary>
|
||||
public static async Task WriteShardedIndexAsync(string repoRoot, string category, Dictionary<int, string> idToPath, int shardSize)
|
||||
{
|
||||
var byBucket = idToPath.GroupBy(kv => kv.Key / shardSize);
|
||||
foreach (var bucket in byBucket)
|
||||
{
|
||||
var ordered = bucket.OrderBy(kv => kv.Key).ToDictionary(kv => kv.Key.ToString(), kv => kv.Value);
|
||||
string path = Path.Combine(repoRoot, "index", $"{category}-by-id", $"{bucket.Key}.json");
|
||||
await DeterministicJsonWriter.WriteIfChangedAsync(path, ordered);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
44
RMuseum/Utils/PublicDataExport/TextFileWriter.cs
Normal file
44
RMuseum/Utils/PublicDataExport/TextFileWriter.cs
Normal file
@ -0,0 +1,44 @@
|
||||
using System.IO;
|
||||
using System.Text;
|
||||
using System.Threading.Tasks;
|
||||
|
||||
namespace RMuseum.Utils.PublicDataExport
|
||||
{
|
||||
/// <summary>
|
||||
/// Same no-op-if-unchanged, LF/UTF-8-no-BOM behavior as <see cref="DeterministicJsonWriter"/>,
|
||||
/// for plain-text files (currently just the generated API.md) rather than JSON.
|
||||
/// </summary>
|
||||
public static class TextFileWriter
|
||||
{
|
||||
private static readonly UTF8Encoding _utf8NoBom = new UTF8Encoding(encoderShouldEmitUTF8Identifier: false);
|
||||
|
||||
public static async Task<bool> WriteIfChangedAsync(string path, string content)
|
||||
{
|
||||
content = content.Replace("\r\n", "\n").TrimEnd('\n') + "\n";
|
||||
byte[] newBytes = _utf8NoBom.GetBytes(content);
|
||||
|
||||
string dir = Path.GetDirectoryName(path);
|
||||
if (!string.IsNullOrEmpty(dir) && !Directory.Exists(dir))
|
||||
{
|
||||
Directory.CreateDirectory(dir);
|
||||
}
|
||||
|
||||
if (File.Exists(path))
|
||||
{
|
||||
byte[] existingBytes = await File.ReadAllBytesAsync(path);
|
||||
if (existingBytes.Length == newBytes.Length)
|
||||
{
|
||||
bool same = true;
|
||||
for (int i = 0; i < existingBytes.Length; i++)
|
||||
{
|
||||
if (existingBytes[i] != newBytes[i]) { same = false; break; }
|
||||
}
|
||||
if (same) return false;
|
||||
}
|
||||
}
|
||||
|
||||
await File.WriteAllBytesAsync(path, newBytes);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
Loading…
Reference in New Issue
Block a user