Fix chapter order, and cover more words without more data

Chapter order: Golestan's دیباچه was rendering below the eight باب because the
listing drew every child category and then every poem. ganjoor.net interleaves
them — a book's preface first, then its chapters, then its remaining poems, so
Hafez reads مقدّمه, five collections, مثنوی, ساقی‌نامه. Ganjoor decides this
with each poem's MixedModeOrder (1 above the chapters, 0 below), which the
exported _cat.json doesn't carry and the live API only exposes one poem at a
time, so prefaces are matched by title for now. Adding MixedModeOrder to the
Poems entries in ganjoor-data would make it exact.

Coverage: I went looking for Arabic and Urdu Wiktionary and measured them
instead of assuming. Against 1,285 distinct words from thirteen poems, Urdu
added nine words and Arabic could reach at most five — the uncovered words
were never Arabic, they were Persian morphology the lookup didn't handle:
enclitic pronouns (آیدت, باشدش, تربتش), negation stacked on a prefix
(برنیاید), and compounds. Deepening the affix chain to two passes takes 87% to
91% with no new data at all, so neither dictionary ships.

Each definition now names its language pair rather than just its source, so a
reader can tell what they are looking at.

Dropped the selection-toolbar lookup: Compose 1.10 stopped routing
SelectionContainer through LocalTextToolbar, so a custom toolbar is never
asked to show, and the replacement in foundation's contextmenu package is
internal. Tapping a word already does the lookup; the note in PoemScreen says
when to revisit.

Verified on an API 36 emulator: Golestan lists دیباچه first, and tapping
نافه‌ای resolves through its affix to ناف with both sources labelled.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Anas Rashid 2026-10-04 17:20:24 +02:00
parent c548e6ec07
commit 8aad086b08
9 changed files with 184 additions and 21 deletions

View File

@ -69,6 +69,18 @@ object Dictionary {
.firstNotNullOfOrNull { direct(database, normalise(it)).ifEmpty { null } }
.orEmpty()
}
// A selection is usually a phrase rather than a word; fall back to its words.
.ifEmpty {
raw.split(' ', '\n', '\r')
.map { normalise(it) }
.filter { it.length > 1 && it != word }
.firstNotNullOfOrNull { part ->
direct(database, part).ifEmpty {
lemmas(database, part).flatMap { direct(database, it) }.ifEmpty { null }
}
}
.orEmpty()
}
}
private fun direct(database: SQLiteDatabase, word: String): List<Definition> =
@ -94,14 +106,29 @@ object Dictionary {
private const val ZWNJ = '‌'
private val SUFFIXES = listOf("ها", "اش", "ش", "م", "ت", "را", "ی", "ان")
// "ال" is the Arabic definite article: poems quote Arabic, so السّاقی has to reach ساقی.
private val PREFIXES = listOf("ال", "می", "بر", "ب")
/**
* Longest first, so تربتش strips شـ rather than matching nothing. These are the endings that
* actually turn up in classical verse: plurals, the object marker, and the enclitic pronouns
* that Persian glues onto a verb or noun — آیدت is آید + ت, باشدش is باشد + ش.
*/
private val SUFFIXES = listOf(
"شان", "تان", "مان", "ها", "اش", "ست", "یم", "ید", "ند", "را", "ش", "م", "ت", "ی", "ان", "ه",
)
/** Candidate stems after stripping one common affix. Order matters: longest affix first. */
internal fun affixes(word: String): List<String> = buildList {
SUFFIXES.forEach { if (word.endsWith(it) && word.length > it.length + 1) add(word.dropLast(it.length)) }
PREFIXES.forEach { if (word.startsWith(it) && word.length > it.length + 1) add(word.drop(it.length)) }
/** "ال" is the Arabic definite article, "ن"/"نمی" negation, "بی" privative. */
private val PREFIXES = listOf("نمی", "ال", "می", "بی", "بر", "ن", "ب")
/**
* Candidate stems, one affix deep and then two — برنیاید is بر + ن + یاید, and a single pass
* would never reach the verb. Ordered so the least mangled candidate is tried first.
*/
internal fun affixes(word: String): List<String> {
fun oneStep(w: String) = buildList {
SUFFIXES.forEach { if (w.endsWith(it) && w.length > it.length + 1) add(w.dropLast(it.length)) }
PREFIXES.forEach { if (w.startsWith(it) && w.length > it.length + 1) add(w.drop(it.length)) }
}
val first = oneStep(word)
return (first + first.flatMap(::oneStep)).distinct()
}
private val HARAKAT = (0x064B..0x0652) + listOf(0x0670, 0x0640) + (0x0610..0x0615)

View File

@ -166,6 +166,37 @@ fun catPath(fullUrl: String) = "poets${fullUrl.trimEnd('/')}/_cat.json"
fun poemPath(fullUrl: String) = "poets${fullUrl.trimEnd('/')}.json"
/** A row in a category listing: either a chapter to open, or a poem to read. */
sealed interface CatEntry {
data class Chapter(val category: Category) : CatEntry
data class Poem(val poem: PoemRef) : CatEntry
}
private val PREFACE_TITLES = listOf("دیباچه", "مقدمه", "سرآغاز", "پیشگفتار", "آغاز")
/**
* Orders a category the way ganjoor.net does: a book's own preface first, then its chapters,
* then whatever other poems sit directly under it. Golestan's دیباچه belongs above the eight
* باب, not below them; Hafez's مقدّمه above his five collections, with مثنوی and ساقی‌نامه after.
*
* Ganjoor decides this with each poem's MixedModeOrder — 1 sorts a poem above the chapters, 0
* below — but that field isn't in the exported `_cat.json`, only on the live API's per-poem
* record, which would be one request per poem.
*
* ponytail: so prefaces are recognised by title instead. Adding MixedModeOrder to the Poems
* entries in ganjoor-data would make this exact; until then a book whose preface is named
* something unusual still lands after its chapters.
*/
fun orderedEntries(category: Category): List<CatEntry> {
val (prefaces, rest) = category.poems.partition { poem ->
val title = normalise(poem.title).trimStart()
PREFACE_TITLES.any { title.startsWith(it) }
}
return prefaces.map(CatEntry::Poem) +
category.childCats.map(CatEntry::Chapter) +
rest.map(CatEntry::Poem)
}
/** One step of a poem's path. [url] is null for the poem itself, which is already open. */
data class Crumb(val label: String, val url: String?)

View File

@ -38,9 +38,11 @@ import androidx.compose.ui.res.stringResource
import androidx.compose.ui.text.style.TextOverflow
import androidx.compose.ui.unit.dp
import com.ganjoor.android.R
import com.ganjoor.android.data.CatEntry
import com.ganjoor.android.data.Downloads
import com.ganjoor.android.data.Ganjoor
import com.ganjoor.android.data.Offline
import com.ganjoor.android.data.orderedEntries
@OptIn(ExperimentalMaterial3Api::class)
@Composable
@ -63,6 +65,8 @@ fun CategoryScreen(
}
}
val entries = remember(cat) { orderedEntries(cat) }
Scaffold(
topBar = {
TopAppBar(
@ -94,12 +98,25 @@ fun CategoryScreen(
cat.description?.takeIf { it.isNotBlank() }?.let { description ->
item { Description(description) }
}
items(cat.childCats, key = { "c${it.id}" }) { child ->
NavRow(child.title, isCategory = true) { onCategory(child.fullUrl) }
}
items(cat.poems, key = { "p${it.id}" }) { poem ->
NavRow(poem.title, isCategory = false, excerpt = excerpts[poem.id]) {
onPoem(poem.fullUrl)
items(
items = entries,
key = { entry ->
when (entry) {
is CatEntry.Chapter -> "c${entry.category.id}"
is CatEntry.Poem -> "p${entry.poem.id}"
}
},
) { entry ->
when (entry) {
is CatEntry.Chapter -> NavRow(entry.category.title, isCategory = true) {
onCategory(entry.category.fullUrl)
}
is CatEntry.Poem -> NavRow(
title = entry.poem.title,
isCategory = false,
excerpt = excerpts[entry.poem.id],
) { onPoem(entry.poem.fullUrl) }
}
}
}

View File

@ -105,8 +105,14 @@ fun PoemScreen(
)
}
) { insets ->
// Free-form selection for copying any span; the per-couplet actions below are for
// saving a passage with the reference attached, which a raw copy would lose.
// Free-form selection for copying any span; tapping a word looks it up, and the
// per-couplet actions save a passage with the reference attached, which a raw copy
// would lose.
//
// ponytail: no dictionary entry in the selection toolbar. Compose 1.10 stopped
// routing SelectionContainer through LocalTextToolbar — a custom TextToolbar is
// simply never asked to show — and the replacement, foundation's contextmenu
// package, is internal. Revisit when that becomes public API.
SelectionContainer {
LazyColumn(
modifier = Modifier.fillMaxSize(),

View File

@ -62,7 +62,7 @@
<string name="about">درباره و پروانه‌ها</string>
<string name="about_intro">گنجور برای اندروید نرم‌افزار آزاد است. شعرها، قلم‌ها و همهٔ کتابخانه‌هایی که این برنامه بر آن‌ها ساخته شده در زیر آمده‌اند؛ برای خواندن متن کامل پروانه روی هر مورد بزنید.</string>
<string name="no_definition">برای این واژه مدخلی یافت نشد.</string>
<string name="source_wiktionary">ویکی‌واژه (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">فرهنگ دانشجو</string>
<string name="source_wiktionary">ویکی‌واژه — فارسی به انگلیسی (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">فرهنگ دانشجو — فارسی به انگلیسی</string>
<string name="dictionary">لغت‌نامه</string>
</resources>

View File

@ -62,7 +62,7 @@
<string name="about">تعارف اور لائسنس</string>
<string name="about_intro">گنجور فار اینڈرائیڈ آزاد سافٹ ویئر ہے۔ کلام، فونٹس اور تمام لائبریریاں ذیل میں درج ہیں؛ مکمل لائسنس پڑھنے کے لیے کسی اندراج پر ٹیپ کریں۔</string>
<string name="no_definition">اس لفظ کا کوئی اندراج نہیں ملا۔</string>
<string name="source_wiktionary">ویکی لغت (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">دانشجو لغت</string>
<string name="source_wiktionary">ویکی لغت — فارسی سے انگریزی (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">دانشجو لغت — فارسی سے انگریزی</string>
<string name="dictionary">لغت نامہ</string>
</resources>

View File

@ -67,7 +67,7 @@
<string name="about">About &amp; licences</string>
<string name="about_intro">Ganjoor for Android is free software. The poems, the fonts and every library it is built on are credited below; tap an entry to read its full licence.</string>
<string name="no_definition">No entry for this word.</string>
<string name="source_wiktionary">Wiktionary (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">Daneshjoo Dictionary</string>
<string name="source_wiktionary">Wiktionary — Persian to English (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">Daneshjoo — Persian to English</string>
<string name="dictionary">Dictionary</string>
</resources>

View File

@ -1,6 +1,10 @@
package com.ganjoor.android
import com.ganjoor.android.data.CatEntry
import com.ganjoor.android.data.Category
import com.ganjoor.android.data.Crumb
import com.ganjoor.android.data.PoemRef
import com.ganjoor.android.data.orderedEntries
import com.ganjoor.android.data.Verse
import com.ganjoor.android.data.breadcrumbs
import com.ganjoor.android.data.parentUrl
@ -125,3 +129,54 @@ class ParentUrlTest {
assertEquals("/hafez", parentUrl("/hafez/ghazal/"))
}
}
class CategoryOrderTest {
private fun cat(chapters: List<String>, poems: List<String>) = Category(
id = 1,
title = "book",
childCats = chapters.mapIndexed { i, t -> Category(id = 100 + i, title = t) },
poems = poems.mapIndexed { i, t -> PoemRef(id = 200 + i, title = t) },
)
private fun titles(category: Category) = orderedEntries(category).map {
when (it) {
is CatEntry.Chapter -> it.category.title
is CatEntry.Poem -> it.poem.title
}
}
@Test
fun `a preface comes before the chapters, as on ganjoor net`() {
// Golestan: دیباچه then the eight باب
val golestan = cat(listOf("باب اول", "باب دوم"), listOf("دیباچه"))
assertEquals(listOf("دیباچه", "باب اول", "باب دوم"), titles(golestan))
}
@Test
fun `other poems stay after the chapters`() {
// Hafez: مقدّمه, then the collections, then مثنوی and ساقی‌نامه
val hafez = cat(
chapters = listOf("غزلیات", "قطعات"),
poems = listOf("مثنوی (الا ای آهوی وحشی)", "ساقی‌نامه", "مقدّمهٔ جمع‌آورندهٔ دیوان حافظ"),
)
assertEquals(
listOf("مقدّمهٔ جمع‌آورندهٔ دیوان حافظ", "غزلیات", "قطعات", "مثنوی (الا ای آهوی وحشی)", "ساقی‌نامه"),
titles(hafez),
)
}
@Test
fun `diacritics in a preface title don't hide it`() {
// مقدّمه carries a shadda the plain spelling doesn't
assertEquals(listOf("مقدّمه", "باب اول"), titles(cat(listOf("باب اول"), listOf("مقدّمه"))))
}
@Test
fun `a category with no poems is left exactly as it is`() {
val masnavi = cat(listOf("دفتر اول", "دفتر دوم", "دفتر سوم"), emptyList())
assertEquals(listOf("دفتر اول", "دفتر دوم", "دفتر سوم"), titles(masnavi))
}
}

View File

@ -79,3 +79,30 @@ class ArabicArticleTest {
assertTrue(affixes("الناس").contains("ناس"))
}
}
class PersianMorphologyTest {
@Test
fun `enclitic pronouns glued onto a verb are stripped`() {
assertTrue(affixes("آیدت").contains("آید"))
assertTrue(affixes("باشدش").contains("باشد"))
assertTrue(affixes("تربتش").contains("تربت"))
}
@Test
fun `a prefix and a negation together still reach the verb`() {
// برنیاید = بر + ن + یاید; one pass would stop at نیاید
assertTrue(affixes("برنیاید").contains("یاید"))
}
@Test
fun `plural and object markers still work`() {
assertTrue(affixes("دلها").contains("دل"))
assertTrue(affixes("مارا").contains("ما"))
}
@Test
fun `stripping never produces a single letter`() {
assertTrue(affixes("شان").none { it.length < 2 })
assertTrue(affixes("بها").none { it.length < 2 })
}
}