diff --git a/GanjooRazor/Areas/Admin/Pages/PublicDataImport.cshtml b/GanjooRazor/Areas/Admin/Pages/PublicDataImport.cshtml
new file mode 100644
index 00000000..5f50dd25
--- /dev/null
+++ b/GanjooRazor/Areas/Admin/Pages/PublicDataImport.cshtml
@@ -0,0 +1,92 @@
+@page
+@model GanjooRazor.Areas.Admin.Pages.PublicDataImportModel
+@addTagHelper *, Microsoft.AspNetCore.Mvc.TagHelpers
+@{
+ Layout = "_AdminLayout";
+ ViewData["Title"] = "درونریزی دادهٔ عمومی گنجور";
+}
+
+
+@if (!string.IsNullOrEmpty(Model.LastMessage))
+{
+ @Model.LastMessage
+}
+
+
+ این ابزار محتوای گنجور (سخنوران، بخشها، شعرها) را از روی
+
مخزن گیت دادهٔ عمومی گنجور
+ (یا یک پوشهٔ محلی که از آن
git clone شده) در پایگاهدادهٔ همین نسخه وارد میکند —
+ راهی سریعتر و سادهتر برای راهاندازی یک فورک یا نسخهٔ محلی کارآمد از گنجور، بهجای واردکردن
+ فایلهای اسکوئللایت بهصورت جداگانه برای هر سخنور.
+
+ هر بار اجرا با خیال راحت قابل تکرار است: فقط دادههایی که هنوز در پایگاهداده نیستند اضافه
+ میشوند و هیچ دادهای بازنویسی یا حذف نمیشود.
+
+ این فرایند در پسزمینه اجرا میشود؛ برای پیگیری وضعیت و درصد پیشرفت به
+
صفحهٔ کارها مراجعه کنید.
+
+
+
+
+
diff --git a/GanjooRazor/Areas/Admin/Pages/PublicDataImport.cshtml.cs b/GanjooRazor/Areas/Admin/Pages/PublicDataImport.cshtml.cs
new file mode 100644
index 00000000..7328c81c
--- /dev/null
+++ b/GanjooRazor/Areas/Admin/Pages/PublicDataImport.cshtml.cs
@@ -0,0 +1,76 @@
+using GanjooRazor.Utils;
+using Microsoft.AspNetCore.Mvc;
+using Microsoft.AspNetCore.Mvc.RazorPages;
+using Newtonsoft.Json;
+using System.Net.Http;
+using System.Text;
+using System.Threading.Tasks;
+
+namespace GanjooRazor.Areas.Admin.Pages
+{
+ ///
+ /// (Re)build local Ganjoor content from the public data export — local git clone or a URL
+ /// (e.g. jsDelivr) — mainly meant to make setting up a fork or local dev copy easier than the
+ /// old per-poet SQLite import.
+ ///
+ [IgnoreAntiforgeryToken(Order = 1001)]
+ public class PublicDataImportModel : PageModel
+ {
+ ///
+ /// last message
+ ///
+ public string LastMessage { get; set; }
+
+ public IActionResult OnGet()
+ {
+ if (string.IsNullOrEmpty(Request.Cookies["Token"]))
+ return Redirect("/");
+
+ LastMessage = "";
+
+ return Page();
+ }
+
+ ///
+ /// trigger the import job (runs in the background — check the Jobs page for progress)
+ ///
+ /// true: location is a URL fetched over HTTP. false: location is a local folder path.
+ /// base URL or local folder path of the exported data tree
+ /// 0 imports every poet; a specific poet id imports only that poet
+ public async Task OnPostImportAsync(bool useHttp, string location, int poetId)
+ {
+ if (string.IsNullOrWhiteSpace(location))
+ {
+ return BadRequest("مسیر یا نشانی نمیتواند خالی باشد.");
+ }
+
+ using (HttpClient secureClient = new HttpClient(new GanjoorReloginHandler(Request, Response)))
+ {
+ if (await GanjoorSessionChecker.PrepareClient(secureClient, Request, Response))
+ {
+ var body = new
+ {
+ useHttp,
+ location,
+ poetId
+ };
+
+ var response = await secureClient.PostAsync
+ (
+ $"{APIRoot.Url}/api/ganjoor/publicdata/import",
+ new StringContent(JsonConvert.SerializeObject(body), Encoding.UTF8, "application/json")
+ );
+
+ if (!response.IsSuccessStatusCode)
+ {
+ return BadRequest(JsonConvert.DeserializeObject(await response.Content.ReadAsStringAsync()));
+ }
+
+ return new OkObjectResult(true);
+ }
+ }
+
+ return new OkObjectResult(false);
+ }
+ }
+}
diff --git a/GanjooRazor/Pages/Shared/_AdminLayout.cshtml b/GanjooRazor/Pages/Shared/_AdminLayout.cshtml
index cbdc3f11..7dee9723 100644
--- a/GanjooRazor/Pages/Shared/_AdminLayout.cshtml
+++ b/GanjooRazor/Pages/Shared/_AdminLayout.cshtml
@@ -456,6 +456,7 @@
کمکهای مالی
هزینهها
کارها
+ درونریزی دادهٔ عمومی
diff --git a/RMuseum/RMuseum.xml b/RMuseum/RMuseum.xml
index c127d051..b60db1d6 100644
--- a/RMuseum/RMuseum.xml
+++ b/RMuseum/RMuseum.xml
@@ -1781,6 +1781,13 @@
+
+
+ (re)build local Ganjoor content from a public data export tree (local folder or HTTP) —
+ intended for local development use, not for production servers
+
+
+
Get user public profile
@@ -11180,6 +11187,28 @@
couplets (virtual) for Masnavi
+
+
+ where to read the exported public data tree from, for StartImportFromPublicDataRepo
+
+
+
+
+ true: fetch over HTTP, Location is a base URL (e.g. jsDelivr).
+ false: read from a local folder, Location is a filesystem path (e.g. a `git clone`).
+
+
+
+
+ base URL (UseHttp=true) or local folder path (UseHttp=false) of the exported data tree
+
+
+
+
+ 0 imports every poet in the export; a specific poet id imports only that poet — handy
+ on a slow connection, or when only one poet's data is needed for local testing
+
+
Root manifest written at the repository root (manifest.json).
@@ -11198,6 +11227,18 @@
timestamps inside poem/cat files, that would defeat deterministic diffs)
+
+
+ number of ids grouped into each id-index shard file (see UrlTemplates.CatIdIndexShard /
+ PoemIdIndexShard). A consumer resolving id X fetches shard file "{X / IdIndexShardSize}.json".
+
+
+
+
+ URL patterns for every file kind in this export, so an app can treat this repo as an
+ API without having to read the export source code. {placeholders} are literal.
+
+
poet.json — biographical data only, no account/user linkage exists on GanjoorPoet at all
@@ -17451,6 +17492,17 @@
+
+
+ (re)build local Ganjoor content (poets/categories/poems/verses/sections + their pages)
+ from a public data export tree, read locally or over HTTP. Safe to re-run — existing
+ entities (by id) are left untouched, only missing ones are added.
+
+ true: fetch over HTTP (location is a base URL). false: read from a local folder (location is a path).
+ base URL or local folder path of the exported data tree
+ 0 imports every poet; a specific id imports only that poet (useful on a slow connection)
+
+
examine site pages for broken links
@@ -20344,6 +20396,9 @@
IGanjoorService implementation
+
+ IGanjoorService implementation
+
@@ -21077,22 +21132,65 @@
+
+
+ how many ids are grouped into each id-index shard file — kept small enough that a
+ shard stays a cheap single fetch, large enough that the id-space doesn't produce an
+ unreasonable number of tiny files. 2000 ids/shard means ~500 shard files for Ganjoor's
+ current poem count.
+
+
- start exporting all published Ganjoor data (poets/categories/poems/verses) to a
+ start exporting all Ganjoor data (poets/categories/poems/verses) belonging to
+ published poets to a
git-tracked JSON tree and pushing it to the configured remote. User-linked tables
(comments, bookmarks, visits, corrections, accounting, ...) are never touched by
this code path — see RMuseum.Models.Ganjoor.PublicExport for the allowlisted shape
of what actually gets written.
-
+
- recursively writes _cat.json for and every published poem directly
- under it, then recurses into published child categories. Returns the number of poems written
- in this subtree (for manifest counts).
+ recursively writes _cat.json for and every poem directly
+ under it, then recurses into child categories. Returns the number of poems written
+ in this subtree (for manifest counts). Only the poet-level Published flag is a real
+ visibility gate in this codebase (see GetPoets) — GanjoorCat.Published and
+ GanjoorPoem.Published are not checked anywhere the live site actually serves content
+ (GetCatByUrl/GetPoemByUrl ignore them entirely), so this export doesn't filter on them
+ either; every category/poem under a published poet is exported.
+
+
+ generates the repo-root API.md every run, so the docs can never drift out of sync with
+ UrlTemplates/IdIndexShardSize in manifest.json. Not hand-edited — if you want to add
+ prose, add it here, not in the generated file.
+
+
+
+
+ Category (and root-poet) pages don't carry a real production page id in the public
+ export — GanjoorPage isn't part of that data set — so this importer mints one
+ deterministically from the category id, kept well clear of any real id range so it can
+ never collide with an actual production GanjoorPage/GanjoorPoem id. Poem pages don't
+ need this: they reuse the poem's own id, matching the convention already used by
+ _ImportSQLiteCatChildren (see GanjoorService-SQLiteImport.cs).
+
+
+
+
+ (Re)builds Ganjoor content — poets, categories, poems, verses, sections, and their
+ GanjoorPage routing entries — from a public data export tree, read either from a local
+ `git clone` or fetched over HTTP. Safe to run against an empty database (bootstrap) or
+ one that already has some content (merge): every entity is looked up by its id first
+ and only inserted if missing, so re-running never duplicates or overwrites anything —
+ including content a developer may have hand-edited locally after a previous import.
+
+ true: fetch over HTTP (location is a base URL). false: read from a local folder (location is a path).
+ base URL or local folder path of the exported data tree
+ 0 imports every poet in the export's manifest; a specific id imports only that poet — useful on a slow connection, or when a developer only needs one poet's data for local testing
+
moderate quoted poems
@@ -23945,6 +24043,30 @@
nothing changed since last run — a normal, expected outcome on most nightly runs).
+
+
+ A bare numeric id (poem id, category id, ...) is meaningless to a static file tree unless
+ something maps it to a path. Writing one giant id->path file doesn't scale to Ganjoor's
+ poem count, so ids are bucketed by id / shardSize into small shard files a client can
+ compute the name of directly — no lookup-before-the-lookup needed.
+
+
+
+
+ Writes every id in to poets-by-id.json (or configured file
+ name) with no sharding — for tables small enough that one file is fine (e.g. poets: a
+ few hundred rows).
+
+
+
+
+ Writes as bucketed shard files under
+ /index/{category}-by-id/{bucket}.json, where
+ bucket = id / shardSize. Only buckets that actually contain ids get a file — an empty
+ bucket produces no request-able file, which is fine since a client only ever asks for
+ the bucket of an id it already has.
+
+
Belt-and-suspenders check on top of the allowlist design: even though the export DTOs
@@ -23965,6 +24087,38 @@
naming the offending type/property if anything trips the checks.
+
+
+ Same no-op-if-unchanged, LF/UTF-8-no-BOM behavior as ,
+ for plain-text files (currently just the generated API.md) rather than JSON.
+
+
+
+
+ Abstracts "read a relative path from the public data export" so the importer doesn't care
+ whether it's reading a local `git clone` or fetching over HTTP from a CDN.
+
+
+
+
+ Returns the file's text content, or null if it doesn't exist at this path
+ (a missing file is a normal, expected outcome — e.g. a leaf category has no children).
+
+
+
+
+ Reads from a local folder — the expected path for a developer who already ran
+ `git clone` on the public data repo, which is both the faster option and the one that
+ doesn't put load on jsDelivr/GitHub for a full-corpus import.
+
+
+
+
+ Fetches over HTTP — point at either the jsDelivr CDN URL
+ (https://cdn.jsdelivr.net/gh/ORG/REPO@main/) or raw.githubusercontent.com. A missing file
+ (404) is treated the same as "doesn't exist", not an error.
+
+
url