diff --git a/.gitea/workflows/sync.yml b/.gitea/workflows/sync.yml index 7836d457a..5d21a7635 100644 --- a/.gitea/workflows/sync.yml +++ b/.gitea/workflows/sync.yml @@ -3,7 +3,7 @@ on: schedule: - cron: "0 3 * * *" # daily 03:00 UTC push: # also run when the sync code changes (bot data commits don't touch these, so no loop) - paths: [wikisource.py, build_index.py, export_diwan.py, update.sh, .gitea/workflows/sync.yml] + paths: [wikisource.py, build_index.py, export_divan.py, update.sh, .gitea/workflows/sync.yml] workflow_dispatch: # "Run workflow" button, Gitea 1.23+ jobs: sync: @@ -11,6 +11,6 @@ jobs: steps: - uses: actions/checkout@v4 - run: | - git config user.name "diwan-bot" - git config user.email "diwan-bot@noreply.git.anasrashid.net" + git config user.name "divan-bot" + git config user.email "divan-bot@noreply.git.anasrashid.net" ./update.sh diff --git a/README.md b/README.md index 3c048372e..6c6d334c0 100644 --- a/README.md +++ b/README.md @@ -1,4 +1,4 @@ -# دیوان · Diwan +# دیوان · Divan A local, searchable SQLite database of **classical Urdu literature, both poetry and prose**: ghazals, nazms, marsiyas, masnavis, letters, dastans and essays by poets and writers such as Mir, Sauda, Dard, Ghalib, Momin, Zauq, Dagh, Anees, Hali, Akbar Allahabadi, Allama Iqbal, Mir Amman, Sir Syed and Nazir Ahmad. Each author entry includes a short introduction. @@ -11,7 +11,7 @@ Everything is in **Urdu script**. Everything is fetched through the official MediaWiki API. -## Database (`diwan.db`) +## Database (`divan.db`) | table | contents | |---|---| @@ -45,16 +45,16 @@ SELECT title, poet_page FROM works_fts WHERE works_fts MATCH 'خودی'; ## Site/API format (Ganjoor-compatible) -The repo root also holds the data in the [ganjoor-data](https://github.com/ganjoor/ganjoor-data) layout (`manifest.json`, `poets/`, `index/`, built by `export_diwan.py`). It works as a static API over jsDelivr with no server: +The repo root also holds the data in the [ganjoor-data](https://github.com/ganjoor/ganjoor-data) layout (`manifest.json`, `poets/`, `index/`, built by `export_divan.py`). It works as a static API over jsDelivr with no server: - https://cdn.jsdelivr.net/gh/anas-rashid/diwan-data@main/manifest.json - https://cdn.jsdelivr.net/gh/anas-rashid/diwan-data@main/poets/p238/_cat.json # Iqbal + https://cdn.jsdelivr.net/gh/anas-rashid/divan-data@main/manifest.json + https://cdn.jsdelivr.net/gh/anas-rashid/divan-data@main/poets/p238/_cat.json # Iqbal It also loads directly into [GanjoorService](https://github.com/ganjoor/GanjoorService) through its "public data import" page: give it the base URL above. Poets are `/p{id}`, categories are a genre slug (`ghazal`, `nazm`, …) or `c{id}`, and poems are `sh{id}`. Ids stay stable across syncs. Poet years are converted to approximate Hijri, following Ganjoor's convention. ## License - **Code:** [MIT](LICENSE). -- **Data** (`diwan.db`, `export/`): the literary works are in the public domain. The compilation is derived from Wikisource and Wikipedia and is shared under **[CC BY-SA 4.0](https://creativecommons.org/licenses/by-sa/4.0/)**. When reusing it, credit *"Urdu Wikisource and Wikipedia contributors"* and share alike. Every record keeps its source `url`. +- **Data** (`divan.db`, `export/`): the literary works are in the public domain. The compilation is derived from Wikisource and Wikipedia and is shared under **[CC BY-SA 4.0](https://creativecommons.org/licenses/by-sa/4.0/)**. When reusing it, credit *"Urdu Wikisource and Wikipedia contributors"* and share alike. Every record keeps its source `url`. Corrections belong upstream: fix the text on Wikisource and re-run the script. diff --git a/build_index.py b/build_index.py index c16380c48..1ff9aa2f0 100644 --- a/build_index.py +++ b/build_index.py @@ -1,9 +1,9 @@ #!/usr/bin/env python3 -"""Build FTS5 full-text search over diwan.db and export JSONL. Safe to re-run.""" +"""Build FTS5 full-text search over divan.db and export JSONL. Safe to re-run.""" import json, os, sqlite3 d = os.path.dirname(os.path.abspath(__file__)) -db = sqlite3.connect(f"{d}/diwan.db") +db = sqlite3.connect(f"{d}/divan.db") db.executescript(""" DROP TABLE IF EXISTS works_fts; CREATE VIRTUAL TABLE works_fts USING fts5(title, poet_page UNINDEXED, kind UNINDEXED, section, text_ur, diff --git a/diwan.db b/divan.db similarity index 99% rename from diwan.db rename to divan.db index 600ae828e..4e109f5df 100644 Binary files a/diwan.db and b/divan.db differ diff --git a/export_diwan.py b/export_divan.py similarity index 95% rename from export_diwan.py rename to export_divan.py index 1a8a3e390..f3de93415 100644 --- a/export_diwan.py +++ b/export_divan.py @@ -1,5 +1,5 @@ #!/usr/bin/env python3 -"""Write diwan.db in the ganjoor-data layout (manifest.json, poets/, index/) so GanjoorService's +"""Write divan.db in the ganjoor-data layout (manifest.json, poets/, index/) so GanjoorService's "public data import" and any ganjoor-data client can read it unchanged. Format: https://github.com/ganjoor/ganjoor-data/blob/main/API.md""" import json, os, re, shutil, sqlite3, time @@ -10,17 +10,17 @@ GENRES = {"غزل": "ghazal", "نظم": "nazm", "رباعی": "rubai", "رباع "مرثیہ": "marsiya", "مثنوی": "masnavi", "قصیدہ": "qasida", "نثر": "nasr", "شاعری": "shaeri", "مضمون": "mazmoon", "خطوط": "khutoot", "سلام": "salam", "نعت": "naat", "حمد": "hamd"} -db = sqlite3.connect(f"{D}/diwan.db") +db = sqlite3.connect(f"{D}/divan.db") # ids are minted once and kept forever, so URLs/ids stay stable across daily syncs -db.execute("CREATE TABLE IF NOT EXISTS diwan_ids(kind TEXT, key TEXT, id INTEGER, PRIMARY KEY(kind, key))") +db.execute("CREATE TABLE IF NOT EXISTS divan_ids(kind TEXT, key TEXT, id INTEGER, PRIMARY KEY(kind, key))") def gid(kind, key): - r = db.execute("SELECT id FROM diwan_ids WHERE kind=? AND key=?", (kind, key)).fetchone() + r = db.execute("SELECT id FROM divan_ids WHERE kind=? AND key=?", (kind, key)).fetchone() if r: return r[0] - n = (db.execute("SELECT max(id) FROM diwan_ids WHERE kind=?", (kind,)).fetchone()[0] or 0) + 1 - db.execute("INSERT INTO diwan_ids VALUES (?,?,?)", (kind, key, n)) + n = (db.execute("SELECT max(id) FROM divan_ids WHERE kind=?", (kind,)).fetchone()[0] or 0) + 1 + db.execute("INSERT INTO divan_ids VALUES (?,?,?)", (kind, key, n)) return n diff --git a/manifest.json b/manifest.json index 19d72d562..e24eda132 100644 --- a/manifest.json +++ b/manifest.json @@ -1,6 +1,6 @@ { "SchemaVersion": 1, - "GeneratedAtUtc": "2026-10-04T22:32:20Z", + "GeneratedAtUtc": "2026-10-08T17:46:52Z", "PoetsCount": 382, "PoemsCount": 11087, "IdIndexShardSize": 2000, diff --git a/update.sh b/update.sh index 762eebaea..8b25bb7e4 100755 --- a/update.sh +++ b/update.sh @@ -4,9 +4,9 @@ set -euo pipefail cd "$(dirname "$0")" python3 wikisource.py python3 build_index.py -python3 export_diwan.py +python3 export_divan.py git add export -if git diff --cached --quiet -- export; then echo "no changes"; exit 0; fi # export/*.jsonl is the change signal; diwan.db bytes change every run +if git diff --cached --quiet -- export; then echo "no changes"; exit 0; fi # export/*.jsonl is the change signal; divan.db bytes change every run git add -A git commit -q -m "data: Wikisource sync $(date -u +%F)" git push -q origin HEAD diff --git a/wikisource.py b/wikisource.py index 7ec4d98ea..83b7d87c6 100644 --- a/wikisource.py +++ b/wikisource.py @@ -1,11 +1,11 @@ #!/usr/bin/env python3 -"""Build diwan.db from Urdu Wikisource (public-domain texts, CC BY-SA 4.0 site) +"""Build divan.db from Urdu Wikisource (public-domain texts, CC BY-SA 4.0 site) plus short poet intros from Urdu Wikipedia. Uses the MediaWiki API, 50 pages per request.""" import json, os, re, sqlite3, time, urllib.parse, urllib.request WS = "https://ur.wikisource.org/w/api.php" -DB = os.path.join(os.path.dirname(os.path.abspath(__file__)), "diwan.db") -UA = "diwan-dataset/0.1 (non-commercial Urdu poetry archive; python-urllib)" +DB = os.path.join(os.path.dirname(os.path.abspath(__file__)), "divan.db") +UA = "divan-dataset/0.1 (non-commercial Urdu poetry archive; python-urllib)" db = sqlite3.connect(DB) db.executescript("""