#498 import public data

This commit is contained in:
Hamid Reza Mohammadi 2026-08-14 16:10:47 +03:30
parent 47f3297cf0
commit 839421ae73
4 changed files with 328 additions and 5 deletions

View File

@ -0,0 +1,92 @@
@page
@model GanjooRazor.Areas.Admin.Pages.PublicDataImportModel
@addTagHelper *, Microsoft.AspNetCore.Mvc.TagHelpers
@{
Layout = "_AdminLayout";
ViewData["Title"] = "درون‌ریزی دادهٔ عمومی گنجور";
}
<div class="up-page-header">
<h1>@ViewData["Title"]</h1>
</div>
@if (!string.IsNullOrEmpty(Model.LastMessage))
{
<div class="up-alert up-alert--info">@Model.LastMessage</div>
}
<div class="up-alert up-alert--info">
این ابزار محتوای گنجور (سخنوران، بخش‌ها، شعرها) را از روی
<a href="https://github.com/ganjoor/ganjoor-data" target="_blank">مخزن گیت دادهٔ عمومی گنجور</a>
(یا یک پوشهٔ محلی که از آن <code>git clone</code> شده) در پایگاه‌دادهٔ همین نسخه وارد می‌کند —
راهی سریع‌تر و ساده‌تر برای راه‌اندازی یک فورک یا نسخهٔ محلی کارآمد از گنجور، به‌جای واردکردن
فایل‌های اسکوئل‌لایت به‌صورت جداگانه برای هر سخنور.
<br />
هر بار اجرا با خیال راحت قابل تکرار است: فقط داده‌هایی که هنوز در پایگاه‌داده نیستند اضافه
می‌شوند و هیچ داده‌ای بازنویسی یا حذف نمی‌شود.
<br />
این فرایند در پس‌زمینه اجرا می‌شود؛ برای پیگیری وضعیت و درصد پیشرفت به
<a asp-area="Admin" asp-page="LongRunningJobs">صفحهٔ کارها</a> مراجعه کنید.
</div>
<script>
async function startPublicDataImport() {
var useHttp = $('#source-http').is(':checked');
var location = $('#location').val();
var poetId = parseInt($('#poet-id').val(), 10) || 0;
if (!location) {
upToast('مسیر پوشه یا نشانی اینترنتی را وارد کنید.', 'error');
return;
}
var confirmMsg = poetId === 0
? 'آیا از درون‌ریزی دادهٔ همهٔ سخنوران اطمینان دارید؟ ممکن است مدتی طول بکشد.'
: 'آیا از درون‌ریزی دادهٔ سخنور با کد ' + poetId + ' اطمینان دارید؟';
var ok = await upConfirm(confirmMsg);
if (!ok) return;
$.ajax({
type: "POST",
url: '?handler=Import',
data: {
useHttp: useHttp,
location: location,
poetId: poetId
},
error: function (e) {
upToast(e.responseText || 'خطا رخ داد.', 'error');
},
success: function () {
upToast('فرایند درون‌ریزی در پس‌زمینه شروع شد. برای پیگیری وضعیت به صفحهٔ کارها مراجعه کنید.', 'success');
},
});
}
</script>
<div style="display:flex;flex-direction:column;gap:var(--up-space-3);max-width:560px;">
<div>
<label style="display:block;margin-bottom:4px;">
<input type="radio" name="source" id="source-local" value="local" checked />
پوشهٔ محلی (که با <code>git clone</code> گرفته‌اید)
</label>
<label style="display:block;">
<input type="radio" name="source" id="source-http" value="http" />
نشانی اینترنتی (مثلاً jsDelivr یا raw.githubusercontent.com)
</label>
</div>
<div>
<label for="location" style="display:block;margin-bottom:4px;">مسیر پوشه یا نشانی:</label>
<input type="text" id="location" style="width:100%;" placeholder="C:\ganjoor-public-data یا https://cdn.jsdelivr.net/gh/ganjoor/ganjoor-data@main/" />
</div>
<div>
<label for="poet-id" style="display:block;margin-bottom:4px;">کد سخنور (۰ برای همهٔ سخنوران):</label>
<input type="number" id="poet-id" value="0" style="width:140px;" />
</div>
<div>
<a role="button" onclick="startPublicDataImport()" class="up-btn up-btn--success">شروع درون‌ریزی</a>
</div>
</div>

View File

@ -0,0 +1,76 @@
using GanjooRazor.Utils;
using Microsoft.AspNetCore.Mvc;
using Microsoft.AspNetCore.Mvc.RazorPages;
using Newtonsoft.Json;
using System.Net.Http;
using System.Text;
using System.Threading.Tasks;
namespace GanjooRazor.Areas.Admin.Pages
{
/// <summary>
/// (Re)build local Ganjoor content from the public data export — local git clone or a URL
/// (e.g. jsDelivr) — mainly meant to make setting up a fork or local dev copy easier than the
/// old per-poet SQLite import.
/// </summary>
[IgnoreAntiforgeryToken(Order = 1001)]
public class PublicDataImportModel : PageModel
{
/// <summary>
/// last message
/// </summary>
public string LastMessage { get; set; }
public IActionResult OnGet()
{
if (string.IsNullOrEmpty(Request.Cookies["Token"]))
return Redirect("/");
LastMessage = "";
return Page();
}
/// <summary>
/// trigger the import job (runs in the background — check the Jobs page for progress)
/// </summary>
/// <param name="useHttp">true: location is a URL fetched over HTTP. false: location is a local folder path.</param>
/// <param name="location">base URL or local folder path of the exported data tree</param>
/// <param name="poetId">0 imports every poet; a specific poet id imports only that poet</param>
public async Task<IActionResult> OnPostImportAsync(bool useHttp, string location, int poetId)
{
if (string.IsNullOrWhiteSpace(location))
{
return BadRequest("مسیر یا نشانی نمی‌تواند خالی باشد.");
}
using (HttpClient secureClient = new HttpClient(new GanjoorReloginHandler(Request, Response)))
{
if (await GanjoorSessionChecker.PrepareClient(secureClient, Request, Response))
{
var body = new
{
useHttp,
location,
poetId
};
var response = await secureClient.PostAsync
(
$"{APIRoot.Url}/api/ganjoor/publicdata/import",
new StringContent(JsonConvert.SerializeObject(body), Encoding.UTF8, "application/json")
);
if (!response.IsSuccessStatusCode)
{
return BadRequest(JsonConvert.DeserializeObject<string>(await response.Content.ReadAsStringAsync()));
}
return new OkObjectResult(true);
}
}
return new OkObjectResult(false);
}
}
}

View File

@ -456,6 +456,7 @@
<a class="dropdown-item" asp-area="Admin" asp-page="Donations">کمک‌های مالی</a>
<a class="dropdown-item" asp-area="Admin" asp-page="Expenses">هزینه‌ها</a>
<a class="dropdown-item" asp-area="Admin" asp-page="LongRunningJobs">کارها</a>
<a class="dropdown-item" asp-area="Admin" asp-page="PublicDataImport">درون‌ریزی دادهٔ عمومی</a>
</div>
</li>
</ul>

View File

@ -1781,6 +1781,13 @@
</summary>
<returns></returns>
</member>
<member name="M:RMuseum.Controllers.GanjoorController.StartImportFromPublicDataRepo(RMuseum.Models.Ganjoor.PublicExport.PublicDataImportRequestDto)">
<summary>
(re)build local Ganjoor content from a public data export tree (local folder or HTTP) —
intended for local development use, not for production servers
</summary>
<returns></returns>
</member>
<member name="M:RMuseum.Controllers.GanjoorController.GetUserPublicProfile(System.Guid)">
<summary>
Get user public profile
@ -11180,6 +11187,28 @@
couplets (virtual) for Masnavi
</summary>
</member>
<member name="T:RMuseum.Models.Ganjoor.PublicExport.PublicDataImportRequestDto">
<summary>
where to read the exported public data tree from, for StartImportFromPublicDataRepo
</summary>
</member>
<member name="P:RMuseum.Models.Ganjoor.PublicExport.PublicDataImportRequestDto.UseHttp">
<summary>
true: fetch over HTTP, Location is a base URL (e.g. jsDelivr).
false: read from a local folder, Location is a filesystem path (e.g. a `git clone`).
</summary>
</member>
<member name="P:RMuseum.Models.Ganjoor.PublicExport.PublicDataImportRequestDto.Location">
<summary>
base URL (UseHttp=true) or local folder path (UseHttp=false) of the exported data tree
</summary>
</member>
<member name="P:RMuseum.Models.Ganjoor.PublicExport.PublicDataImportRequestDto.PoetId">
<summary>
0 imports every poet in the export; a specific poet id imports only that poet — handy
on a slow connection, or when only one poet's data is needed for local testing
</summary>
</member>
<member name="T:RMuseum.Models.Ganjoor.PublicExport.PublicExportManifestDto">
<summary>
Root manifest written at the repository root (manifest.json).
@ -11198,6 +11227,18 @@
timestamps inside poem/cat files, that would defeat deterministic diffs)
</summary>
</member>
<member name="P:RMuseum.Models.Ganjoor.PublicExport.PublicExportManifestDto.IdIndexShardSize">
<summary>
number of ids grouped into each id-index shard file (see UrlTemplates.CatIdIndexShard /
PoemIdIndexShard). A consumer resolving id X fetches shard file "{X / IdIndexShardSize}.json".
</summary>
</member>
<member name="P:RMuseum.Models.Ganjoor.PublicExport.PublicExportManifestDto.UrlTemplates">
<summary>
URL patterns for every file kind in this export, so an app can treat this repo as an
API without having to read the export source code. {placeholders} are literal.
</summary>
</member>
<member name="T:RMuseum.Models.Ganjoor.PublicExport.PoetPublicDto">
<summary>
poet.json — biographical data only, no account/user linkage exists on GanjoorPoet at all
@ -17451,6 +17492,17 @@
</summary>
<returns></returns>
</member>
<member name="M:RMuseum.Services.IGanjoorService.StartImportFromPublicDataRepo(System.Boolean,System.String,System.Int32)">
<summary>
(re)build local Ganjoor content (poets/categories/poems/verses/sections + their pages)
from a public data export tree, read locally or over HTTP. Safe to re-run — existing
entities (by id) are left untouched, only missing ones are added.
</summary>
<param name="useHttp">true: fetch over HTTP (location is a base URL). false: read from a local folder (location is a path).</param>
<param name="location">base URL or local folder path of the exported data tree</param>
<param name="poetId">0 imports every poet; a specific id imports only that poet (useful on a slow connection)</param>
<returns></returns>
</member>
<member name="M:RMuseum.Services.IGanjoorService.HealthCheckContents">
<summary>
examine site pages for broken links
@ -20344,6 +20396,9 @@
<summary>
IGanjoorService implementation
</summary>
<summary>
IGanjoorService implementation
</summary>
</member>
<member name="M:RMuseum.Services.Implementation.GanjoorService.SwitchCoupletBookmark(System.Guid,System.Int32,System.Int32)">
<summary>
@ -21077,22 +21132,65 @@
<param name="metre"></param>
<returns></returns>
</member>
<member name="F:RMuseum.Services.Implementation.GanjoorService.IdIndexShardSize">
<summary>
how many ids are grouped into each id-index shard file — kept small enough that a
shard stays a cheap single fetch, large enough that the id-space doesn't produce an
unreasonable number of tiny files. 2000 ids/shard means ~500 shard files for Ganjoor's
current poem count.
</summary>
</member>
<member name="M:RMuseum.Services.Implementation.GanjoorService.StartBatchExportPublicGitData">
<summary>
start exporting all published Ganjoor data (poets/categories/poems/verses) to a
start exporting all Ganjoor data (poets/categories/poems/verses) belonging to
published poets to a
git-tracked JSON tree and pushing it to the configured remote. User-linked tables
(comments, bookmarks, visits, corrections, accounting, ...) are never touched by
this code path — see RMuseum.Models.Ganjoor.PublicExport for the allowlisted shape
of what actually gets written.
</summary>
</member>
<member name="M:RMuseum.Services.Implementation.GanjoorService.ExportCatTreeToJson(RMuseum.DbContext.RMuseumDbContext,System.String,RMuseum.Models.Ganjoor.GanjoorCat)">
<member name="M:RMuseum.Services.Implementation.GanjoorService.ExportCatTreeToJson(RMuseum.DbContext.RMuseumDbContext,System.String,RMuseum.Models.Ganjoor.GanjoorCat,System.Collections.Generic.Dictionary{System.Int32,System.String},System.Collections.Generic.Dictionary{System.Int32,System.String})">
<summary>
recursively writes _cat.json for <paramref name="cat"/> and every published poem directly
under it, then recurses into published child categories. Returns the number of poems written
in this subtree (for manifest counts).
recursively writes _cat.json for <paramref name="cat"/> and every poem directly
under it, then recurses into child categories. Returns the number of poems written
in this subtree (for manifest counts). Only the poet-level Published flag is a real
visibility gate in this codebase (see GetPoets) — GanjoorCat.Published and
GanjoorPoem.Published are not checked anywhere the live site actually serves content
(GetCatByUrl/GetPoemByUrl ignore them entirely), so this export doesn't filter on them
either; every category/poem under a published poet is exported.
</summary>
</member>
<member name="M:RMuseum.Services.Implementation.GanjoorService.BuildApiMarkdown(RMuseum.Models.Ganjoor.PublicExport.PublicExportManifestDto)">
<summary>
generates the repo-root API.md every run, so the docs can never drift out of sync with
UrlTemplates/IdIndexShardSize in manifest.json. Not hand-edited — if you want to add
prose, add it here, not in the generated file.
</summary>
</member>
<member name="M:RMuseum.Services.Implementation.GanjoorService.SyntheticCatPageId(System.Int32)">
<summary>
Category (and root-poet) pages don't carry a real production page id in the public
export — GanjoorPage isn't part of that data set — so this importer mints one
deterministically from the category id, kept well clear of any real id range so it can
never collide with an actual production GanjoorPage/GanjoorPoem id. Poem pages don't
need this: they reuse the poem's own id, matching the convention already used by
_ImportSQLiteCatChildren (see GanjoorService-SQLiteImport.cs).
</summary>
</member>
<member name="M:RMuseum.Services.Implementation.GanjoorService.StartImportFromPublicDataRepo(System.Boolean,System.String,System.Int32)">
<summary>
(Re)builds Ganjoor content — poets, categories, poems, verses, sections, and their
GanjoorPage routing entries — from a public data export tree, read either from a local
`git clone` or fetched over HTTP. Safe to run against an empty database (bootstrap) or
one that already has some content (merge): every entity is looked up by its id first
and only inserted if missing, so re-running never duplicates or overwrites anything —
including content a developer may have hand-edited locally after a previous import.
</summary>
<param name="useHttp">true: fetch over HTTP (location is a base URL). false: read from a local folder (location is a path).</param>
<param name="location">base URL or local folder path of the exported data tree</param>
<param name="poetId">0 imports every poet in the export's manifest; a specific id imports only that poet — useful on a slow connection, or when a developer only needs one poet's data for local testing</param>
</member>
<member name="M:RMuseum.Services.Implementation.GanjoorService.ModerateGanjoorQuotedPoemAsync(RMuseum.Models.Ganjoor.ViewModels.GanjoorQuotedPoemModerationViewModel,System.Guid)">
<summary>
moderate quoted poems
@ -23945,6 +24043,30 @@
nothing changed since last run — a normal, expected outcome on most nightly runs).
</summary>
</member>
<member name="T:RMuseum.Utils.PublicDataExport.IdIndexWriter">
<summary>
A bare numeric id (poem id, category id, ...) is meaningless to a static file tree unless
something maps it to a path. Writing one giant id-&gt;path file doesn't scale to Ganjoor's
poem count, so ids are bucketed by <c>id / shardSize</c> into small shard files a client can
compute the name of directly — no lookup-before-the-lookup needed.
</summary>
</member>
<member name="M:RMuseum.Utils.PublicDataExport.IdIndexWriter.WriteFlatIndexAsync(System.String,System.String,System.Collections.Generic.Dictionary{System.Int32,System.String})">
<summary>
Writes every id in <paramref name="idToPath"/> to poets-by-id.json (or configured file
name) with no sharding — for tables small enough that one file is fine (e.g. poets: a
few hundred rows).
</summary>
</member>
<member name="M:RMuseum.Utils.PublicDataExport.IdIndexWriter.WriteShardedIndexAsync(System.String,System.String,System.Collections.Generic.Dictionary{System.Int32,System.String},System.Int32)">
<summary>
Writes <paramref name="idToPath"/> as bucketed shard files under
<paramref name="repoRoot"/>/index/{category}-by-id/{bucket}.json, where
bucket = id / shardSize. Only buckets that actually contain ids get a file — an empty
bucket produces no request-able file, which is fine since a client only ever asks for
the bucket of an id it already has.
</summary>
</member>
<member name="T:RMuseum.Utils.PublicDataExport.PublicExportSafetyGuard">
<summary>
Belt-and-suspenders check on top of the allowlist design: even though the export DTOs
@ -23965,6 +24087,38 @@
naming the offending type/property if anything trips the checks.
</summary>
</member>
<member name="T:RMuseum.Utils.PublicDataExport.TextFileWriter">
<summary>
Same no-op-if-unchanged, LF/UTF-8-no-BOM behavior as <see cref="T:RMuseum.Utils.PublicDataExport.DeterministicJsonWriter"/>,
for plain-text files (currently just the generated API.md) rather than JSON.
</summary>
</member>
<member name="T:RMuseum.Utils.PublicDataImport.IPublicDataSource">
<summary>
Abstracts "read a relative path from the public data export" so the importer doesn't care
whether it's reading a local `git clone` or fetching over HTTP from a CDN.
</summary>
</member>
<member name="M:RMuseum.Utils.PublicDataImport.IPublicDataSource.ReadTextAsync(System.String)">
<summary>
Returns the file's text content, or null if it doesn't exist at this path
(a missing file is a normal, expected outcome — e.g. a leaf category has no children).
</summary>
</member>
<member name="T:RMuseum.Utils.PublicDataImport.LocalFileSystemPublicDataSource">
<summary>
Reads from a local folder — the expected path for a developer who already ran
`git clone` on the public data repo, which is both the faster option and the one that
doesn't put load on jsDelivr/GitHub for a full-corpus import.
</summary>
</member>
<member name="T:RMuseum.Utils.PublicDataImport.HttpPublicDataSource">
<summary>
Fetches over HTTP — point <paramref name="baseUrl"/> at either the jsDelivr CDN URL
(https://cdn.jsdelivr.net/gh/ORG/REPO@main/) or raw.githubusercontent.com. A missing file
(404) is treated the same as "doesn't exist", not an error.
</summary>
</member>
<member name="P:RMuseum.WebServiceUrl.Url">
<summary>
url