diff --git a/GanjooRazor/Areas/Admin/Pages/PublicDataImport.cshtml b/GanjooRazor/Areas/Admin/Pages/PublicDataImport.cshtml new file mode 100644 index 00000000..5f50dd25 --- /dev/null +++ b/GanjooRazor/Areas/Admin/Pages/PublicDataImport.cshtml @@ -0,0 +1,92 @@ +@page +@model GanjooRazor.Areas.Admin.Pages.PublicDataImportModel +@addTagHelper *, Microsoft.AspNetCore.Mvc.TagHelpers +@{ + Layout = "_AdminLayout"; + ViewData["Title"] = "درون‌ریزی دادهٔ عمومی گنجور"; +} +
+

@ViewData["Title"]

+
+ +@if (!string.IsNullOrEmpty(Model.LastMessage)) +{ +
@Model.LastMessage
+} + +
+ این ابزار محتوای گنجور (سخنوران، بخش‌ها، شعرها) را از روی + مخزن گیت دادهٔ عمومی گنجور + (یا یک پوشهٔ محلی که از آن git clone شده) در پایگاه‌دادهٔ همین نسخه وارد می‌کند — + راهی سریع‌تر و ساده‌تر برای راه‌اندازی یک فورک یا نسخهٔ محلی کارآمد از گنجور، به‌جای واردکردن + فایل‌های اسکوئل‌لایت به‌صورت جداگانه برای هر سخنور. +
+ هر بار اجرا با خیال راحت قابل تکرار است: فقط داده‌هایی که هنوز در پایگاه‌داده نیستند اضافه + می‌شوند و هیچ داده‌ای بازنویسی یا حذف نمی‌شود. +
+ این فرایند در پس‌زمینه اجرا می‌شود؛ برای پیگیری وضعیت و درصد پیشرفت به + صفحهٔ کارها مراجعه کنید. +
+ + + +
+
+ + +
+ +
+ + +
+ +
+ + +
+ +
+ شروع درون‌ریزی +
+
diff --git a/GanjooRazor/Areas/Admin/Pages/PublicDataImport.cshtml.cs b/GanjooRazor/Areas/Admin/Pages/PublicDataImport.cshtml.cs new file mode 100644 index 00000000..7328c81c --- /dev/null +++ b/GanjooRazor/Areas/Admin/Pages/PublicDataImport.cshtml.cs @@ -0,0 +1,76 @@ +using GanjooRazor.Utils; +using Microsoft.AspNetCore.Mvc; +using Microsoft.AspNetCore.Mvc.RazorPages; +using Newtonsoft.Json; +using System.Net.Http; +using System.Text; +using System.Threading.Tasks; + +namespace GanjooRazor.Areas.Admin.Pages +{ + /// + /// (Re)build local Ganjoor content from the public data export — local git clone or a URL + /// (e.g. jsDelivr) — mainly meant to make setting up a fork or local dev copy easier than the + /// old per-poet SQLite import. + /// + [IgnoreAntiforgeryToken(Order = 1001)] + public class PublicDataImportModel : PageModel + { + /// + /// last message + /// + public string LastMessage { get; set; } + + public IActionResult OnGet() + { + if (string.IsNullOrEmpty(Request.Cookies["Token"])) + return Redirect("/"); + + LastMessage = ""; + + return Page(); + } + + /// + /// trigger the import job (runs in the background — check the Jobs page for progress) + /// + /// true: location is a URL fetched over HTTP. false: location is a local folder path. + /// base URL or local folder path of the exported data tree + /// 0 imports every poet; a specific poet id imports only that poet + public async Task OnPostImportAsync(bool useHttp, string location, int poetId) + { + if (string.IsNullOrWhiteSpace(location)) + { + return BadRequest("مسیر یا نشانی نمی‌تواند خالی باشد."); + } + + using (HttpClient secureClient = new HttpClient(new GanjoorReloginHandler(Request, Response))) + { + if (await GanjoorSessionChecker.PrepareClient(secureClient, Request, Response)) + { + var body = new + { + useHttp, + location, + poetId + }; + + var response = await secureClient.PostAsync + ( + $"{APIRoot.Url}/api/ganjoor/publicdata/import", + new StringContent(JsonConvert.SerializeObject(body), Encoding.UTF8, "application/json") + ); + + if (!response.IsSuccessStatusCode) + { + return BadRequest(JsonConvert.DeserializeObject(await response.Content.ReadAsStringAsync())); + } + + return new OkObjectResult(true); + } + } + + return new OkObjectResult(false); + } + } +} diff --git a/GanjooRazor/Pages/Shared/_AdminLayout.cshtml b/GanjooRazor/Pages/Shared/_AdminLayout.cshtml index cbdc3f11..7dee9723 100644 --- a/GanjooRazor/Pages/Shared/_AdminLayout.cshtml +++ b/GanjooRazor/Pages/Shared/_AdminLayout.cshtml @@ -456,6 +456,7 @@ کمک‌های مالی هزینه‌ها کارها + درون‌ریزی دادهٔ عمومی diff --git a/RMuseum/RMuseum.xml b/RMuseum/RMuseum.xml index c127d051..b60db1d6 100644 --- a/RMuseum/RMuseum.xml +++ b/RMuseum/RMuseum.xml @@ -1781,6 +1781,13 @@ + + + (re)build local Ganjoor content from a public data export tree (local folder or HTTP) — + intended for local development use, not for production servers + + + Get user public profile @@ -11180,6 +11187,28 @@ couplets (virtual) for Masnavi + + + where to read the exported public data tree from, for StartImportFromPublicDataRepo + + + + + true: fetch over HTTP, Location is a base URL (e.g. jsDelivr). + false: read from a local folder, Location is a filesystem path (e.g. a `git clone`). + + + + + base URL (UseHttp=true) or local folder path (UseHttp=false) of the exported data tree + + + + + 0 imports every poet in the export; a specific poet id imports only that poet — handy + on a slow connection, or when only one poet's data is needed for local testing + + Root manifest written at the repository root (manifest.json). @@ -11198,6 +11227,18 @@ timestamps inside poem/cat files, that would defeat deterministic diffs) + + + number of ids grouped into each id-index shard file (see UrlTemplates.CatIdIndexShard / + PoemIdIndexShard). A consumer resolving id X fetches shard file "{X / IdIndexShardSize}.json". + + + + + URL patterns for every file kind in this export, so an app can treat this repo as an + API without having to read the export source code. {placeholders} are literal. + + poet.json — biographical data only, no account/user linkage exists on GanjoorPoet at all @@ -17451,6 +17492,17 @@ + + + (re)build local Ganjoor content (poets/categories/poems/verses/sections + their pages) + from a public data export tree, read locally or over HTTP. Safe to re-run — existing + entities (by id) are left untouched, only missing ones are added. + + true: fetch over HTTP (location is a base URL). false: read from a local folder (location is a path). + base URL or local folder path of the exported data tree + 0 imports every poet; a specific id imports only that poet (useful on a slow connection) + + examine site pages for broken links @@ -20344,6 +20396,9 @@ IGanjoorService implementation + + IGanjoorService implementation + @@ -21077,22 +21132,65 @@ + + + how many ids are grouped into each id-index shard file — kept small enough that a + shard stays a cheap single fetch, large enough that the id-space doesn't produce an + unreasonable number of tiny files. 2000 ids/shard means ~500 shard files for Ganjoor's + current poem count. + + - start exporting all published Ganjoor data (poets/categories/poems/verses) to a + start exporting all Ganjoor data (poets/categories/poems/verses) belonging to + published poets to a git-tracked JSON tree and pushing it to the configured remote. User-linked tables (comments, bookmarks, visits, corrections, accounting, ...) are never touched by this code path — see RMuseum.Models.Ganjoor.PublicExport for the allowlisted shape of what actually gets written. - + - recursively writes _cat.json for and every published poem directly - under it, then recurses into published child categories. Returns the number of poems written - in this subtree (for manifest counts). + recursively writes _cat.json for and every poem directly + under it, then recurses into child categories. Returns the number of poems written + in this subtree (for manifest counts). Only the poet-level Published flag is a real + visibility gate in this codebase (see GetPoets) — GanjoorCat.Published and + GanjoorPoem.Published are not checked anywhere the live site actually serves content + (GetCatByUrl/GetPoemByUrl ignore them entirely), so this export doesn't filter on them + either; every category/poem under a published poet is exported. + + + generates the repo-root API.md every run, so the docs can never drift out of sync with + UrlTemplates/IdIndexShardSize in manifest.json. Not hand-edited — if you want to add + prose, add it here, not in the generated file. + + + + + Category (and root-poet) pages don't carry a real production page id in the public + export — GanjoorPage isn't part of that data set — so this importer mints one + deterministically from the category id, kept well clear of any real id range so it can + never collide with an actual production GanjoorPage/GanjoorPoem id. Poem pages don't + need this: they reuse the poem's own id, matching the convention already used by + _ImportSQLiteCatChildren (see GanjoorService-SQLiteImport.cs). + + + + + (Re)builds Ganjoor content — poets, categories, poems, verses, sections, and their + GanjoorPage routing entries — from a public data export tree, read either from a local + `git clone` or fetched over HTTP. Safe to run against an empty database (bootstrap) or + one that already has some content (merge): every entity is looked up by its id first + and only inserted if missing, so re-running never duplicates or overwrites anything — + including content a developer may have hand-edited locally after a previous import. + + true: fetch over HTTP (location is a base URL). false: read from a local folder (location is a path). + base URL or local folder path of the exported data tree + 0 imports every poet in the export's manifest; a specific id imports only that poet — useful on a slow connection, or when a developer only needs one poet's data for local testing + moderate quoted poems @@ -23945,6 +24043,30 @@ nothing changed since last run — a normal, expected outcome on most nightly runs). + + + A bare numeric id (poem id, category id, ...) is meaningless to a static file tree unless + something maps it to a path. Writing one giant id->path file doesn't scale to Ganjoor's + poem count, so ids are bucketed by id / shardSize into small shard files a client can + compute the name of directly — no lookup-before-the-lookup needed. + + + + + Writes every id in to poets-by-id.json (or configured file + name) with no sharding — for tables small enough that one file is fine (e.g. poets: a + few hundred rows). + + + + + Writes as bucketed shard files under + /index/{category}-by-id/{bucket}.json, where + bucket = id / shardSize. Only buckets that actually contain ids get a file — an empty + bucket produces no request-able file, which is fine since a client only ever asks for + the bucket of an id it already has. + + Belt-and-suspenders check on top of the allowlist design: even though the export DTOs @@ -23965,6 +24087,38 @@ naming the offending type/property if anything trips the checks. + + + Same no-op-if-unchanged, LF/UTF-8-no-BOM behavior as , + for plain-text files (currently just the generated API.md) rather than JSON. + + + + + Abstracts "read a relative path from the public data export" so the importer doesn't care + whether it's reading a local `git clone` or fetching over HTTP from a CDN. + + + + + Returns the file's text content, or null if it doesn't exist at this path + (a missing file is a normal, expected outcome — e.g. a leaf category has no children). + + + + + Reads from a local folder — the expected path for a developer who already ran + `git clone` on the public data repo, which is both the faster option and the one that + doesn't put load on jsDelivr/GitHub for a full-corpus import. + + + + + Fetches over HTTP — point at either the jsDelivr CDN URL + (https://cdn.jsdelivr.net/gh/ORG/REPO@main/) or raw.githubusercontent.com. A missing file + (404) is treated the same as "doesn't exist", not an error. + + url