Fix chapter order, and cover more words without more data
Chapter order: Golestan's دیباچه was rendering below the eight باب because the listing drew every child category and then every poem. ganjoor.net interleaves them — a book's preface first, then its chapters, then its remaining poems, so Hafez reads مقدّمه, five collections, مثنوی, ساقینامه. Ganjoor decides this with each poem's MixedModeOrder (1 above the chapters, 0 below), which the exported _cat.json doesn't carry and the live API only exposes one poem at a time, so prefaces are matched by title for now. Adding MixedModeOrder to the Poems entries in ganjoor-data would make it exact. Coverage: I went looking for Arabic and Urdu Wiktionary and measured them instead of assuming. Against 1,285 distinct words from thirteen poems, Urdu added nine words and Arabic could reach at most five — the uncovered words were never Arabic, they were Persian morphology the lookup didn't handle: enclitic pronouns (آیدت, باشدش, تربتش), negation stacked on a prefix (برنیاید), and compounds. Deepening the affix chain to two passes takes 87% to 91% with no new data at all, so neither dictionary ships. Each definition now names its language pair rather than just its source, so a reader can tell what they are looking at. Dropped the selection-toolbar lookup: Compose 1.10 stopped routing SelectionContainer through LocalTextToolbar, so a custom toolbar is never asked to show, and the replacement in foundation's contextmenu package is internal. Tapping a word already does the lookup; the note in PoemScreen says when to revisit. Verified on an API 36 emulator: Golestan lists دیباچه first, and tapping نافهای resolves through its affix to ناف with both sources labelled. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
1 parent
c548e6ec07
commit
8aad086b08
9 files changed
+184
-21
No files matched your search
@@ -1,6 +1,10 @@
|
||||
package com.ganjoor.android
|
||||
|
||||
import com.ganjoor.android.data.CatEntry
|
||||
import com.ganjoor.android.data.Category
|
||||
import com.ganjoor.android.data.Crumb
|
||||
import com.ganjoor.android.data.PoemRef
|
||||
import com.ganjoor.android.data.orderedEntries
|
||||
import com.ganjoor.android.data.Verse
|
||||
import com.ganjoor.android.data.breadcrumbs
|
||||
import com.ganjoor.android.data.parentUrl
|
||||
@@ -125,3 +129,54 @@ class ParentUrlTest {
|
||||
assertEquals("/hafez", parentUrl("/hafez/ghazal/"))
|
||||
}
|
||||
}
|
||||
|
||||
class CategoryOrderTest {
|
||||
private fun cat(chapters: List<String>, poems: List<String>) = Category(
|
||||
id = 1,
|
||||
title = "book",
|
||||
childCats = chapters.mapIndexed { i, t -> Category(id = 100 + i, title = t) },
|
||||
poems = poems.mapIndexed { i, t -> PoemRef(id = 200 + i, title = t) },
|
||||
)
|
||||
|
||||
private fun titles(category: Category) = orderedEntries(category).map {
|
||||
when (it) {
|
||||
is CatEntry.Chapter -> it.category.title
|
||||
is CatEntry.Poem -> it.poem.title
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `a preface comes before the chapters, as on ganjoor net`() {
|
||||
// Golestan: دیباچه then the eight باب
|
||||
val golestan = cat(listOf("باب اول", "باب دوم"), listOf("دیباچه"))
|
||||
|
||||
assertEquals(listOf("دیباچه", "باب اول", "باب دوم"), titles(golestan))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `other poems stay after the chapters`() {
|
||||
// Hafez: مقدّمه, then the collections, then مثنوی and ساقینامه
|
||||
val hafez = cat(
|
||||
chapters = listOf("غزلیات", "قطعات"),
|
||||
poems = listOf("مثنوی (الا ای آهوی وحشی)", "ساقینامه", "مقدّمهٔ جمعآورندهٔ دیوان حافظ"),
|
||||
)
|
||||
|
||||
assertEquals(
|
||||
listOf("مقدّمهٔ جمعآورندهٔ دیوان حافظ", "غزلیات", "قطعات", "مثنوی (الا ای آهوی وحشی)", "ساقینامه"),
|
||||
titles(hafez),
|
||||
)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `diacritics in a preface title don't hide it`() {
|
||||
// مقدّمه carries a shadda the plain spelling doesn't
|
||||
assertEquals(listOf("مقدّمه", "باب اول"), titles(cat(listOf("باب اول"), listOf("مقدّمه"))))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `a category with no poems is left exactly as it is`() {
|
||||
val masnavi = cat(listOf("دفتر اول", "دفتر دوم", "دفتر سوم"), emptyList())
|
||||
|
||||
assertEquals(listOf("دفتر اول", "دفتر دوم", "دفتر سوم"), titles(masnavi))
|
||||
}
|
||||
}
|
||||
@@ -79,3 +79,30 @@ class ArabicArticleTest {
|
||||
assertTrue(affixes("الناس").contains("ناس"))
|
||||
}
|
||||
}
|
||||
|
||||
class PersianMorphologyTest {
|
||||
@Test
|
||||
fun `enclitic pronouns glued onto a verb are stripped`() {
|
||||
assertTrue(affixes("آیدت").contains("آید"))
|
||||
assertTrue(affixes("باشدش").contains("باشد"))
|
||||
assertTrue(affixes("تربتش").contains("تربت"))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `a prefix and a negation together still reach the verb`() {
|
||||
// برنیاید = بر + ن + یاید; one pass would stop at نیاید
|
||||
assertTrue(affixes("برنیاید").contains("یاید"))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `plural and object markers still work`() {
|
||||
assertTrue(affixes("دلها").contains("دل"))
|
||||
assertTrue(affixes("مارا").contains("ما"))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `stripping never produces a single letter`() {
|
||||
assertTrue(affixes("شان").none { it.length < 2 })
|
||||
assertTrue(affixes("بها").none { it.length < 2 })
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user