From b3f4e527f16698388cc2b14245b3b0726a00c3d1 Mon Sep 17 00:00:00 2001 From: anas Date: Wed, 7 Oct 2026 17:08:19 +0200 Subject: [PATCH] Look up the word, not the word and the comma after it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A tapped word was taken as the longest run of characters in U+0600–U+06FF, but that block is not only letters. The Arabic comma ، semicolon ؛ question mark ؟ full stop ۔ and both sets of Indic digits all live inside it, so the run ran straight through them: tapping دستم in «ز دستم، صاحب‌دلان» looked up «دستم،», which no dictionary carries and which the near-word search cannot rescue either, since the punctuation counts against every candidate's letter overlap. The run now keeps only letters, the marks that sit on them, and the joiner, decided by Unicode category rather than a hand-kept list of code points. Harakat and tatweel stay in: they sit inside a word — منِ is one word — and normalise() strips them before the lookup. Latin punctuation already fell outside the block. The range is also what the verse highlights, so the comma is no longer painted as part of the tapped word. Co-Authored-By: Claude Opus 5 (1M context) --- .../com/ganjoor/android/data/Dictionary.kt | 21 ++++++++++++- .../com/ganjoor/android/DictionaryTest.kt | 31 +++++++++++++++++++ 2 files changed, 51 insertions(+), 1 deletion(-) diff --git a/app/src/main/java/com/ganjoor/android/data/Dictionary.kt b/app/src/main/java/com/ganjoor/android/data/Dictionary.kt index 4e0e963..d4d5284 100644 --- a/app/src/main/java/com/ganjoor/android/data/Dictionary.kt +++ b/app/src/main/java/com/ganjoor/android/data/Dictionary.kt @@ -279,4 +279,23 @@ internal fun wordRangeAt(text: String, index: Int): IntRange? { return (start..end).takeIf { end - start + 1 > 1 } } -private fun Char.isWordChar() = this in '؀'..'ۿ' || this == ZWNJ +/** + * Letters, the marks that sit on them, and the joiner — but not the punctuation a line of verse + * is pointed with, nor its digits. + * + * The Arabic block holds far more than letters: the comma ، the semicolon ؛ the question mark ؟ + * the full stop ۔ and both sets of Indic digits all live inside it. Spanning the whole block swept + * them into the word, so tapping دستم in «ز دستم، صاحب‌دلان» looked up «دستم،», which no dictionary + * carries and no near-word search rescues. + * + * Harakat stay in: they sit *inside* a word — منِ is one word — and normalise() strips them before + * the lookup anyway. Tatweel stays for the same reason, stretching a letter without breaking it. + */ +private fun Char.isWordChar() = + this == ZWNJ || (this in '؀'..'ۿ' && category in WORD_CATEGORIES) + +private val WORD_CATEGORIES = setOf( + CharCategory.OTHER_LETTER, // the letters themselves + CharCategory.NON_SPACING_MARK, // harakat, which sit on a letter + CharCategory.MODIFIER_LETTER, // tatweel, which stretches one +) diff --git a/app/src/test/java/com/ganjoor/android/DictionaryTest.kt b/app/src/test/java/com/ganjoor/android/DictionaryTest.kt index 8f7f09f..f52c556 100644 --- a/app/src/test/java/com/ganjoor/android/DictionaryTest.kt +++ b/app/src/test/java/com/ganjoor/android/DictionaryTest.kt @@ -134,4 +134,35 @@ class LetterOverlapTest { fun `an empty word never matches`() { assertEquals(0f, letterOverlap("", "عشق"), 0.001f) } + + @Test + fun `an arabic comma is not part of the word before it`() { + val line = "دل می‌رود ز دستم، صاحب‌دلان خدا را" + assertEquals("دستم", wordAt(line, line.indexOf("دستم") + 1)) + } + + @Test + fun `a question mark is not part of the word before it`() { + val line = "صلاح کار کجا و من خراب کجا؟" + assertEquals("کجا", wordAt(line, line.lastIndexOf("کجا") + 1)) + } + + @Test + fun `a full stop and a semicolon do not join the word`() { + assertEquals("یافت", wordAt("نخواهی یافت۔", 8)) + assertEquals("برخیز", wordAt("شرطه برخیز؛ باد", 7)) + } + + @Test + fun `digits are not part of a word`() { + assertEquals("غزل", wordAt("غزل۳", 1)) + assertEquals("غزل", wordAt("غزل١٢", 1)) + } + + @Test + fun `harakat keep a word whole`() { + // منِ is one word: the kasra sits inside it, and normalise strips it before the lookup. + val line = "و منِ خراب" + assertEquals("منِ", wordAt(line, line.indexOf("من") + 1)) + } }