Add the Urdu dictionaries, and suggest near words when nothing matches

Two Urdu sources join the Persian ones. Wiktionary's Urdu extract gives Urdu
headwords glossed in English, and Urdu Wiktionary itself gives definitions
written in Urdu — the only source here that does. The latter is thin, about
3,100 usable entries out of 31,000 pages since many are stubs, but for a word
it carries an Urdu reader is better served by it than by a translation into
English: عشق comes back as شدید جذبۂ محبت، گہری چاہت، محبت، پریم، پیار.

Every definition now names the dictionary and its language pair, and lays out
in the direction its own script reads, so an Urdu definition is right-aligned
beside a left-aligned English one.

When nothing matches, the sheet offers near words ranked by how many letters
they share with what was looked up, drawn from an index range scan on the
leading letters rather than a scan of the whole table. خودکامی, which has no
entry, offers خودکامه — the lemma it wants.

Measured honestly: the Urdu sources add little coverage over Persian — nine
words from the English-glossed extract, two from Urdu Wiktionary, against
1,285 from thirteen poems. They are here because an Urdu reader wants Urdu,
not because they widen the net.

The suggestion ranking is verified against the built database rather than only
on device: خودکامی → خودکامه, شیرازی → شیراز, مشکلها → مشکل. On an API 36
emulator all four sources answer عشق with their labels.

Known rough edge: affix stripping across four languages can mislead. ناولها
reaches ناول, the Urdu for "novel", which is not what Hafez meant. The matched
headword is always shown, so it is visible rather than silent.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Anas Rashid 2026-10-04 17:29:38 +02:00
parent 8aad086b08
commit c0cb6a8ab6
11 changed files with 259 additions and 24 deletions

Binary file not shown.

View File

@ -17,7 +17,7 @@ data class Definition(val word: String, val gloss: String, val source: String)
*/
object Dictionary {
/** Guards against a copy interrupted half-way leaving an unopenable file behind. */
private const val ASSET_BYTES = 22_298_624L
private const val ASSET_BYTES = 27742208L
// Not a .gz: the build packager silently gunzips those and drops the extension, which left
// the asset under a different name than the code was opening.
@ -95,6 +95,47 @@ object Dictionary {
}
}
/**
* Headwords that look like [raw], for when nothing matched exactly — a misread letter, an
* unusual spelling, or a word the dictionary simply spells differently.
*
* Candidates come from an index range scan on the first letters, then are ranked by how
* many letters they share with the query. ponytail: shared letters rather than an edit
* distance, which would need the whole table scanned to be worth the extra precision.
*/
suspend fun suggest(raw: String, limit: Int = 6): List<String> = withContext(Dispatchers.IO) {
val database = open() ?: return@withContext emptyList()
val word = normalise(raw).takeIf { it.length > 1 } ?: return@withContext emptyList()
// Widen the prefix until there is something to rank, but never scan the whole table.
val candidates = generateSequence(minOf(3, word.length - 1)) { (it - 1).takeIf { n -> n >= 1 } }
.map { prefixLength -> byPrefix(database, word.take(prefixLength)) }
.firstOrNull { it.size >= 3 }
?: return@withContext emptyList()
candidates
.asSequence()
.filter { it.first != word }
.map { (normalised, display) -> display to letterOverlap(word, normalised) }
.filter { it.second > 0.45f }
.sortedByDescending { it.second }
.map { it.first }
.distinct()
.take(limit)
.toList()
}
/** Index range scan: everything whose normalised form starts with [prefix]. */
private fun byPrefix(database: SQLiteDatabase, prefix: String): List<Pair<String, String>> =
database.rawQuery(
"SELECT DISTINCT word, display FROM entry WHERE word >= ? AND word < ? LIMIT 400",
arrayOf(prefix, prefix + '\uFFFF'),
).use { cursor ->
buildList {
while (cursor.moveToNext()) add(cursor.getString(0) to cursor.getString(1))
}
}
private fun lemmas(database: SQLiteDatabase, form: String): List<String> =
database.rawQuery(
"SELECT lemma FROM form WHERE form = ? LIMIT 6",
@ -152,6 +193,17 @@ internal fun normalise(text: String, keepZwnj: Boolean = false): String {
return (if (keepZwnj) folded else folded.replace(ZWNJ.toString(), "")).trim()
}
/**
* How much of two words' letters coincide — the overlapping letters counted against the longer
* word, so مشکل and مشکلها score high while a word that merely starts the same does not.
*/
internal fun letterOverlap(a: String, b: String): Float {
if (a.isEmpty() || b.isEmpty()) return 0f
val remaining = b.toMutableList()
val shared = a.count { remaining.remove(it) }
return shared.toFloat() / maxOf(a.length, b.length)
}
/** The whole word surrounding [index], for turning a tap into something to look up. */
internal fun wordAt(text: String, index: Int): String? {
if (text.isEmpty()) return null

View File

@ -64,11 +64,17 @@ private val CREDITS = listOf(
),
Credit(
"Wiktionary",
"Wiktionary contributors — the word definitions, and the inflected-form index that " +
"finds a conjugated verb's dictionary entry",
"Wiktionary contributors — Persian and Urdu definitions, and the inflected-form " +
"index that finds a conjugated verb's dictionary entry",
"CC BY-SA 3.0. The bundled dictionary is therefore also CC BY-SA 3.0.",
url = "https://en.wiktionary.org",
),
Credit(
"Urdu Wiktionary",
"Urdu Wiktionary contributors — the definitions written in Urdu rather than English",
"CC BY-SA 3.0. The bundled dictionary is therefore also CC BY-SA 3.0.",
url = "https://ur.wiktionary.org",
),
Credit(
"Daneshjoo Dictionary",
"Layered under Wiktionary for the words it doesn't carry",

View File

@ -7,17 +7,21 @@ import androidx.compose.foundation.layout.navigationBarsPadding
import androidx.compose.foundation.layout.padding
import androidx.compose.foundation.rememberScrollState
import androidx.compose.foundation.verticalScroll
import androidx.compose.foundation.layout.FlowRow
import androidx.compose.material3.CircularProgressIndicator
import androidx.compose.material3.ExperimentalMaterial3Api
import androidx.compose.material3.HorizontalDivider
import androidx.compose.material3.MaterialTheme
import androidx.compose.material3.ModalBottomSheet
import androidx.compose.material3.SuggestionChip
import androidx.compose.material3.Text
import androidx.compose.runtime.Composable
import androidx.compose.runtime.CompositionLocalProvider
import androidx.compose.runtime.getValue
import androidx.compose.runtime.mutableStateOf
import androidx.compose.runtime.produceState
import androidx.compose.runtime.remember
import androidx.compose.runtime.setValue
import androidx.compose.ui.Modifier
import androidx.compose.ui.platform.LocalLayoutDirection
import androidx.compose.ui.res.stringResource
@ -34,12 +38,39 @@ private fun LeftToRight(content: @Composable () -> Unit) {
CompositionLocalProvider(LocalLayoutDirection provides LayoutDirection.Ltr, content = content)
}
/** Lays a definition out the way its own script reads. */
@Composable
private fun InDirectionOf(text: String, content: @Composable () -> Unit) {
val arabicScript = text.count { it in '\u0600'..'\u06FF' }
val latin = text.count { it in 'A'..'Z' || it in 'a'..'z' }
CompositionLocalProvider(
LocalLayoutDirection provides
if (arabicScript > latin) LayoutDirection.Rtl else LayoutDirection.Ltr,
content = content,
)
}
/** Which dictionary answered, and in which language pair. */
private fun sourceLabel(source: String) = when (source) {
"wiktionary-fa" -> R.string.source_wiktionary
"wiktionary-ur" -> R.string.source_wiktionary_ur
"urwiktionary" -> R.string.source_urwiktionary
else -> R.string.source_daneshjoo
}
/** What the dictionary knows about a tapped word. */
@OptIn(ExperimentalMaterial3Api::class)
@Composable
fun WordSheet(word: String, onDismiss: () -> Unit) {
val prefs = LocalSettings.current.value
val definitions by produceState<List<Definition>?>(null, word) { value = Dictionary.lookup(word) }
// A suggestion replaces what is being looked up, so the sheet can be followed like a trail.
var current by remember(word) { mutableStateOf(word) }
val definitions by produceState<List<Definition>?>(null, current) {
value = Dictionary.lookup(current)
}
val suggestions by produceState(emptyList<String>(), current, definitions) {
value = if (definitions?.isEmpty() == true) Dictionary.suggest(current) else emptyList()
}
ModalBottomSheet(onDismissRequest = onDismiss) {
Column(
@ -52,7 +83,7 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
) {
// The headword in the reading font, at reading size: it is a line of poetry, after all.
Text(
text = word,
text = current,
style = readingStyle(prefs.font, prefs.fontSize, prefs.fontWeight.weight),
modifier = Modifier.fillMaxWidth(),
)
@ -61,7 +92,10 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
when {
definitions == null -> CircularProgressIndicator(Modifier.padding(vertical = 16.dp))
definitions!!.isEmpty() -> LeftToRight {
definitions!!.isEmpty() -> Column(
verticalArrangement = Arrangement.spacedBy(8.dp),
) {
LeftToRight {
Text(
text = stringResource(R.string.no_definition),
style = MaterialTheme.typography.bodyMedium,
@ -69,30 +103,41 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
modifier = Modifier.fillMaxWidth(),
)
}
if (suggestions.isNotEmpty()) {
Text(
text = stringResource(R.string.did_you_mean),
style = MaterialTheme.typography.titleSmall,
color = MaterialTheme.colorScheme.primary,
)
FlowRow(horizontalArrangement = Arrangement.spacedBy(8.dp)) {
suggestions.forEach { suggestion ->
SuggestionChip(
onClick = { current = suggestion },
label = { Text(suggestion) },
)
}
}
}
}
else -> definitions!!.forEach { definition ->
Column(Modifier.fillMaxWidth()) {
// The headword actually matched, which may be the lemma rather than the
// word as it appears in the line. Persian, so it stays right-to-left.
if (definition.word != word) {
if (definition.word != current) {
Text(
text = definition.word,
style = MaterialTheme.typography.titleSmall,
color = MaterialTheme.colorScheme.primary,
)
}
// The definitions are English; right-aligning them reads badly.
LeftToRight {
// Most definitions are English and right-aligning them reads badly;
// the Urdu ones are right-to-left like the rest of the app.
InDirectionOf(definition.gloss) {
Column(Modifier.fillMaxWidth()) {
Text(definition.gloss, style = MaterialTheme.typography.bodyMedium)
Text(
text = stringResource(
if (definition.source == "wiktionary") {
R.string.source_wiktionary
} else {
R.string.source_daneshjoo
}
),
text = stringResource(sourceLabel(definition.source)),
style = MaterialTheme.typography.labelSmall,
color = MaterialTheme.colorScheme.onSurfaceVariant,
)

View File

@ -65,4 +65,7 @@
<string name="source_wiktionary">ویکی‌واژه — فارسی به انگلیسی (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">فرهنگ دانشجو — فارسی به انگلیسی</string>
<string name="dictionary">لغت‌نامه</string>
<string name="source_wiktionary_ur">ویکی‌واژه — اردو به انگلیسی (CC BY-SA 3.0)</string>
<string name="source_urwiktionary">ویکی‌واژهٔ اردو — اردو به اردو (CC BY-SA 3.0)</string>
<string name="did_you_mean">واژه‌های نزدیک</string>
</resources>

View File

@ -65,4 +65,7 @@
<string name="source_wiktionary">ویکی لغت — فارسی سے انگریزی (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">دانشجو لغت — فارسی سے انگریزی</string>
<string name="dictionary">لغت نامہ</string>
<string name="source_wiktionary_ur">ویکی لغت — اردو سے انگریزی (CC BY-SA 3.0)</string>
<string name="source_urwiktionary">اردو ویکی لغت — اردو سے اردو (CC BY-SA 3.0)</string>
<string name="did_you_mean">ملتے جلتے الفاظ</string>
</resources>

View File

@ -70,4 +70,7 @@
<string name="source_wiktionary">Wiktionary — Persian to English (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">Daneshjoo — Persian to English</string>
<string name="dictionary">Dictionary</string>
<string name="source_wiktionary_ur">Wiktionary — Urdu to English (CC BY-SA 3.0)</string>
<string name="source_urwiktionary">Urdu Wiktionary — Urdu to Urdu (CC BY-SA 3.0)</string>
<string name="did_you_mean">Similar words</string>
</resources>

View File

@ -1,6 +1,7 @@
package com.ganjoor.android
import com.ganjoor.android.data.affixes
import com.ganjoor.android.data.letterOverlap
import com.ganjoor.android.data.normalise
import com.ganjoor.android.data.wordAt
import org.junit.Assert.assertEquals
@ -106,3 +107,31 @@ class PersianMorphologyTest {
assertTrue(affixes("بها").none { it.length < 2 })
}
}
class LetterOverlapTest {
@Test
fun `an identical word overlaps completely`() {
assertEquals(1f, letterOverlap("عشق", "عشق"), 0.001f)
}
@Test
fun `a suffixed form still scores high against its stem`() {
assertTrue(letterOverlap("مشکل", "مشکلها") > 0.6f)
}
@Test
fun `sharing only a first letter scores low`() {
assertTrue(letterOverlap("عشق", "عبادتگاه") < 0.4f)
}
@Test
fun `letters are counted once each, not by presence alone`() {
// ااا against ا shares one letter, not three
assertEquals(1f / 3f, letterOverlap("ااا", "ا"), 0.001f)
}
@Test
fun `an empty word never matches`() {
assertEquals(0f, letterOverlap("", "عشق"), 0.001f)
}
}

View File

@ -19,6 +19,25 @@ bundled rather than merely linked.
Nothing in Ganjoor's repositories restricts AI-assisted use.
## Dictionary
Four sources, each row in the database tagged with the one it came from so the app can name it
and so any of them can be dropped without rebuilding the others.
| Source | Direction | Licence |
|---|---|---|
| [Wiktionary](https://en.wiktionary.org) | Persian → English | CC BY-SA 3.0 |
| [Wiktionary](https://en.wiktionary.org) | Urdu → English | CC BY-SA 3.0 |
| [Urdu Wiktionary](https://ur.wiktionary.org) | Urdu → Urdu | CC BY-SA 3.0 |
| [Daneshjoo](https://github.com/0xdolan/Daneshjoo) | Persian → English | Repository states MIT |
Because three of the four are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**.
Attribution is shown in the app beside every definition. See [`../tools/README.md`](../tools/README.md)
to rebuild it.
The Daneshjoo entry is worth a caveat: the repository states MIT, but the underlying lexicon is
a published Iranian dictionary, so that relicensing is worth verifying before relying on it.
## Fonts
| Font | Copyright | Licence | File |

View File

@ -30,6 +30,20 @@ tables, etymology templates, IPA and descendants. The build keeps the definition
form→lemma index (149,589 pairs) and drops the rest. That index is what resolves conjugated
verbs: `افتاد → افتادن`, `بگشاید → گشودن`, `دانند → دانستن`.
## Urdu sources
```sh
curl -L -o ur.jsonl https://kaikki.org/dictionary/Urdu/kaikki.org-dictionary-Urdu.jsonl
curl -L -o urwikt.xml.bz2 \
https://dumps.wikimedia.org/urwiktionary/latest/urwiktionary-latest-pages-articles.xml.bz2
bunzip2 -k urwikt.xml.bz2
```
The first gives Urdu headwords glossed in English. The second is Urdu Wiktionary itself, the only
source here whose definitions are written **in Urdu** — thin (around 3,100 usable entries out of
31,000 pages, many being stubs), but for a word it does carry an Urdu reader is better served by
it than by a translation into English.
## Licences
Each row carries its `source`, so attribution stays accurate and either source can be dropped

View File

@ -1,5 +1,11 @@
"""Builds the app's dictionary from Wiktionary (CC BY-SA 3.0) and Daneshjoo (MIT).
Three sources, each row tagged so the app can say which one answered and in which language:
wiktionary-fa Persian headwords, English definitions
daneshjoo Persian headwords, English definitions
wiktionary-ur Urdu headwords, English definitions
Wiktionary's export is 93 MB of linguistic metadata around 0.94 MB of definitions, so this keeps
the definitions and the form->lemma index and discards the rest. The `source` column is what
keeps the attribution honest and lets either source be dropped later.
@ -36,13 +42,32 @@ for line in open('fa.jsonl', encoding='utf-8'):
if gs:
pos = e.get('pos') or ''
gloss = '; '.join(dict.fromkeys(gs))[:600]
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary'))
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa'))
for f in e.get('forms', []):
t = f.get('form')
if t and t != word and not t.startswith('-') and len(t) > 1:
forms.add((normalise(t), normalise(word)))
print(f"wiktionary: {len(entries)} entries, {len(forms)} forms")
print(f"wiktionary-fa: {len(entries)} entries, {len(forms)} forms")
# Urdu shares a great deal of vocabulary with Persian, so these entries answer words the
# Persian sources miss. Headwords are Urdu; the definitions are still English.
n_fa = len(entries)
for line in open('ur.jsonl', encoding='utf-8'):
try: e = json.loads(line)
except Exception: continue
word = e.get('word')
if not word: continue
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
if gs:
pos = e.get('pos') or ''
gloss = '; '.join(dict.fromkeys(gs))[:600]
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur'))
for f in e.get('forms', []):
t = f.get('form')
if t and t != word and not t.startswith('-') and len(t) > 1:
forms.add((normalise(t), normalise(word)))
print(f"wiktionary-ur: {len(entries) - n_fa} entries, {len(forms)} forms total")
tag = re.compile(r'<[^>]+>')
n0 = len(entries)
@ -58,6 +83,42 @@ for k, v in MDX('daneshjoo.mdx').items():
entries.append((normalise(word), word, txt[:600], 'daneshjoo'))
print(f"daneshjoo : {len(entries) - n0} entries")
# Urdu Wiktionary, the only source here whose definitions are written in Urdu rather than
# English. Thin — a few thousand usable entries, many pages being stubs — but for a word it
# does carry, an Urdu reader is better served by it than by a translation into English.
if os.path.exists('urwikt.xml'):
import html as _html
raw = open('urwikt.xml', encoding='utf-8', errors='ignore').read()
n_ur = len(entries)
for title, ns, body in re.findall(
r'<title>(.*?)</title>.*?<ns>(\d+)</ns>.*?<text[^>]*>(.*?)</text>', raw, re.S
):
if ns != '0' or not re.match(r'^[\u0600-\u06FF]', title):
continue
t = re.sub(r'\{\{[^}]*\}\}', ' ', body)
t = re.sub(r'\[\[([^\]|]*\|)?([^\]]*)\]\]', r'\2', t)
t = re.sub(r"'{2,}|<[^>]+>", '', _html.unescape(t))
section = re.search(r'==\s*معانی\s*==(.*?)(?:\n==|\Z)', t, re.S)
lines = (section.group(1) if section else
'\n'.join(l.strip(' #') for l in t.split('\n') if l.strip().startswith('#')))
kept = []
for line in lines.split('\n'):
line = re.sub(r'^\d+\.\s*', '', line.strip())
# ؎ introduces a verse citation, and "ref"/a year starts the source note; the
# definition itself is what comes before either.
if line.startswith('؎') or re.match(r'^\(?\s*\d{3,4}ء', line):
break
line = re.split(r'\bref\b|؎', line)[0].strip()
if not line or not re.search(r'[\u0600-\u06FF]', line):
continue
kept.append(line)
if len(' '.join(kept)) > 220:
break
gloss = ' '.join(kept).strip()[:300]
if len(gloss) > 3:
entries.append((normalise(title), title, gloss, 'urwiktionary'))
print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)")
c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries)
c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms))
c.executescript("""