Dictionary coverage, pronunciation, and chapter order #1
Binary file not shown.
@ -17,7 +17,7 @@ data class Definition(val word: String, val gloss: String, val source: String)
|
||||
*/
|
||||
object Dictionary {
|
||||
/** Guards against a copy interrupted half-way leaving an unopenable file behind. */
|
||||
private const val ASSET_BYTES = 22_298_624L
|
||||
private const val ASSET_BYTES = 27742208L
|
||||
|
||||
// Not a .gz: the build packager silently gunzips those and drops the extension, which left
|
||||
// the asset under a different name than the code was opening.
|
||||
@ -95,6 +95,47 @@ object Dictionary {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Headwords that look like [raw], for when nothing matched exactly — a misread letter, an
|
||||
* unusual spelling, or a word the dictionary simply spells differently.
|
||||
*
|
||||
* Candidates come from an index range scan on the first letters, then are ranked by how
|
||||
* many letters they share with the query. ponytail: shared letters rather than an edit
|
||||
* distance, which would need the whole table scanned to be worth the extra precision.
|
||||
*/
|
||||
suspend fun suggest(raw: String, limit: Int = 6): List<String> = withContext(Dispatchers.IO) {
|
||||
val database = open() ?: return@withContext emptyList()
|
||||
val word = normalise(raw).takeIf { it.length > 1 } ?: return@withContext emptyList()
|
||||
|
||||
// Widen the prefix until there is something to rank, but never scan the whole table.
|
||||
val candidates = generateSequence(minOf(3, word.length - 1)) { (it - 1).takeIf { n -> n >= 1 } }
|
||||
.map { prefixLength -> byPrefix(database, word.take(prefixLength)) }
|
||||
.firstOrNull { it.size >= 3 }
|
||||
?: return@withContext emptyList()
|
||||
|
||||
candidates
|
||||
.asSequence()
|
||||
.filter { it.first != word }
|
||||
.map { (normalised, display) -> display to letterOverlap(word, normalised) }
|
||||
.filter { it.second > 0.45f }
|
||||
.sortedByDescending { it.second }
|
||||
.map { it.first }
|
||||
.distinct()
|
||||
.take(limit)
|
||||
.toList()
|
||||
}
|
||||
|
||||
/** Index range scan: everything whose normalised form starts with [prefix]. */
|
||||
private fun byPrefix(database: SQLiteDatabase, prefix: String): List<Pair<String, String>> =
|
||||
database.rawQuery(
|
||||
"SELECT DISTINCT word, display FROM entry WHERE word >= ? AND word < ? LIMIT 400",
|
||||
arrayOf(prefix, prefix + '\uFFFF'),
|
||||
).use { cursor ->
|
||||
buildList {
|
||||
while (cursor.moveToNext()) add(cursor.getString(0) to cursor.getString(1))
|
||||
}
|
||||
}
|
||||
|
||||
private fun lemmas(database: SQLiteDatabase, form: String): List<String> =
|
||||
database.rawQuery(
|
||||
"SELECT lemma FROM form WHERE form = ? LIMIT 6",
|
||||
@ -152,6 +193,17 @@ internal fun normalise(text: String, keepZwnj: Boolean = false): String {
|
||||
return (if (keepZwnj) folded else folded.replace(ZWNJ.toString(), "")).trim()
|
||||
}
|
||||
|
||||
/**
|
||||
* How much of two words' letters coincide — the overlapping letters counted against the longer
|
||||
* word, so مشکل and مشکلها score high while a word that merely starts the same does not.
|
||||
*/
|
||||
internal fun letterOverlap(a: String, b: String): Float {
|
||||
if (a.isEmpty() || b.isEmpty()) return 0f
|
||||
val remaining = b.toMutableList()
|
||||
val shared = a.count { remaining.remove(it) }
|
||||
return shared.toFloat() / maxOf(a.length, b.length)
|
||||
}
|
||||
|
||||
/** The whole word surrounding [index], for turning a tap into something to look up. */
|
||||
internal fun wordAt(text: String, index: Int): String? {
|
||||
if (text.isEmpty()) return null
|
||||
|
||||
@ -64,11 +64,17 @@ private val CREDITS = listOf(
|
||||
),
|
||||
Credit(
|
||||
"Wiktionary",
|
||||
"Wiktionary contributors — the word definitions, and the inflected-form index that " +
|
||||
"finds a conjugated verb's dictionary entry",
|
||||
"Wiktionary contributors — Persian and Urdu definitions, and the inflected-form " +
|
||||
"index that finds a conjugated verb's dictionary entry",
|
||||
"CC BY-SA 3.0. The bundled dictionary is therefore also CC BY-SA 3.0.",
|
||||
url = "https://en.wiktionary.org",
|
||||
),
|
||||
Credit(
|
||||
"Urdu Wiktionary",
|
||||
"Urdu Wiktionary contributors — the definitions written in Urdu rather than English",
|
||||
"CC BY-SA 3.0. The bundled dictionary is therefore also CC BY-SA 3.0.",
|
||||
url = "https://ur.wiktionary.org",
|
||||
),
|
||||
Credit(
|
||||
"Daneshjoo Dictionary",
|
||||
"Layered under Wiktionary for the words it doesn't carry",
|
||||
|
||||
@ -7,17 +7,21 @@ import androidx.compose.foundation.layout.navigationBarsPadding
|
||||
import androidx.compose.foundation.layout.padding
|
||||
import androidx.compose.foundation.rememberScrollState
|
||||
import androidx.compose.foundation.verticalScroll
|
||||
import androidx.compose.foundation.layout.FlowRow
|
||||
import androidx.compose.material3.CircularProgressIndicator
|
||||
import androidx.compose.material3.ExperimentalMaterial3Api
|
||||
import androidx.compose.material3.HorizontalDivider
|
||||
import androidx.compose.material3.MaterialTheme
|
||||
import androidx.compose.material3.ModalBottomSheet
|
||||
import androidx.compose.material3.SuggestionChip
|
||||
import androidx.compose.material3.Text
|
||||
import androidx.compose.runtime.Composable
|
||||
import androidx.compose.runtime.CompositionLocalProvider
|
||||
import androidx.compose.runtime.getValue
|
||||
import androidx.compose.runtime.mutableStateOf
|
||||
import androidx.compose.runtime.produceState
|
||||
import androidx.compose.runtime.remember
|
||||
import androidx.compose.runtime.setValue
|
||||
import androidx.compose.ui.Modifier
|
||||
import androidx.compose.ui.platform.LocalLayoutDirection
|
||||
import androidx.compose.ui.res.stringResource
|
||||
@ -34,12 +38,39 @@ private fun LeftToRight(content: @Composable () -> Unit) {
|
||||
CompositionLocalProvider(LocalLayoutDirection provides LayoutDirection.Ltr, content = content)
|
||||
}
|
||||
|
||||
/** Lays a definition out the way its own script reads. */
|
||||
@Composable
|
||||
private fun InDirectionOf(text: String, content: @Composable () -> Unit) {
|
||||
val arabicScript = text.count { it in '\u0600'..'\u06FF' }
|
||||
val latin = text.count { it in 'A'..'Z' || it in 'a'..'z' }
|
||||
CompositionLocalProvider(
|
||||
LocalLayoutDirection provides
|
||||
if (arabicScript > latin) LayoutDirection.Rtl else LayoutDirection.Ltr,
|
||||
content = content,
|
||||
)
|
||||
}
|
||||
|
||||
/** Which dictionary answered, and in which language pair. */
|
||||
private fun sourceLabel(source: String) = when (source) {
|
||||
"wiktionary-fa" -> R.string.source_wiktionary
|
||||
"wiktionary-ur" -> R.string.source_wiktionary_ur
|
||||
"urwiktionary" -> R.string.source_urwiktionary
|
||||
else -> R.string.source_daneshjoo
|
||||
}
|
||||
|
||||
/** What the dictionary knows about a tapped word. */
|
||||
@OptIn(ExperimentalMaterial3Api::class)
|
||||
@Composable
|
||||
fun WordSheet(word: String, onDismiss: () -> Unit) {
|
||||
val prefs = LocalSettings.current.value
|
||||
val definitions by produceState<List<Definition>?>(null, word) { value = Dictionary.lookup(word) }
|
||||
// A suggestion replaces what is being looked up, so the sheet can be followed like a trail.
|
||||
var current by remember(word) { mutableStateOf(word) }
|
||||
val definitions by produceState<List<Definition>?>(null, current) {
|
||||
value = Dictionary.lookup(current)
|
||||
}
|
||||
val suggestions by produceState(emptyList<String>(), current, definitions) {
|
||||
value = if (definitions?.isEmpty() == true) Dictionary.suggest(current) else emptyList()
|
||||
}
|
||||
|
||||
ModalBottomSheet(onDismissRequest = onDismiss) {
|
||||
Column(
|
||||
@ -52,7 +83,7 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
|
||||
) {
|
||||
// The headword in the reading font, at reading size: it is a line of poetry, after all.
|
||||
Text(
|
||||
text = word,
|
||||
text = current,
|
||||
style = readingStyle(prefs.font, prefs.fontSize, prefs.fontWeight.weight),
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
)
|
||||
@ -61,38 +92,52 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
|
||||
when {
|
||||
definitions == null -> CircularProgressIndicator(Modifier.padding(vertical = 16.dp))
|
||||
|
||||
definitions!!.isEmpty() -> LeftToRight {
|
||||
Text(
|
||||
text = stringResource(R.string.no_definition),
|
||||
style = MaterialTheme.typography.bodyMedium,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
)
|
||||
definitions!!.isEmpty() -> Column(
|
||||
verticalArrangement = Arrangement.spacedBy(8.dp),
|
||||
) {
|
||||
LeftToRight {
|
||||
Text(
|
||||
text = stringResource(R.string.no_definition),
|
||||
style = MaterialTheme.typography.bodyMedium,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
)
|
||||
}
|
||||
if (suggestions.isNotEmpty()) {
|
||||
Text(
|
||||
text = stringResource(R.string.did_you_mean),
|
||||
style = MaterialTheme.typography.titleSmall,
|
||||
color = MaterialTheme.colorScheme.primary,
|
||||
)
|
||||
FlowRow(horizontalArrangement = Arrangement.spacedBy(8.dp)) {
|
||||
suggestions.forEach { suggestion ->
|
||||
SuggestionChip(
|
||||
onClick = { current = suggestion },
|
||||
label = { Text(suggestion) },
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
else -> definitions!!.forEach { definition ->
|
||||
Column(Modifier.fillMaxWidth()) {
|
||||
// The headword actually matched, which may be the lemma rather than the
|
||||
// word as it appears in the line. Persian, so it stays right-to-left.
|
||||
if (definition.word != word) {
|
||||
if (definition.word != current) {
|
||||
Text(
|
||||
text = definition.word,
|
||||
style = MaterialTheme.typography.titleSmall,
|
||||
color = MaterialTheme.colorScheme.primary,
|
||||
)
|
||||
}
|
||||
// The definitions are English; right-aligning them reads badly.
|
||||
LeftToRight {
|
||||
// Most definitions are English and right-aligning them reads badly;
|
||||
// the Urdu ones are right-to-left like the rest of the app.
|
||||
InDirectionOf(definition.gloss) {
|
||||
Column(Modifier.fillMaxWidth()) {
|
||||
Text(definition.gloss, style = MaterialTheme.typography.bodyMedium)
|
||||
Text(
|
||||
text = stringResource(
|
||||
if (definition.source == "wiktionary") {
|
||||
R.string.source_wiktionary
|
||||
} else {
|
||||
R.string.source_daneshjoo
|
||||
}
|
||||
),
|
||||
text = stringResource(sourceLabel(definition.source)),
|
||||
style = MaterialTheme.typography.labelSmall,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
|
||||
@ -65,4 +65,7 @@
|
||||
<string name="source_wiktionary">ویکیواژه — فارسی به انگلیسی (CC BY-SA 3.0)</string>
|
||||
<string name="source_daneshjoo">فرهنگ دانشجو — فارسی به انگلیسی</string>
|
||||
<string name="dictionary">لغتنامه</string>
|
||||
<string name="source_wiktionary_ur">ویکیواژه — اردو به انگلیسی (CC BY-SA 3.0)</string>
|
||||
<string name="source_urwiktionary">ویکیواژهٔ اردو — اردو به اردو (CC BY-SA 3.0)</string>
|
||||
<string name="did_you_mean">واژههای نزدیک</string>
|
||||
</resources>
|
||||
|
||||
@ -65,4 +65,7 @@
|
||||
<string name="source_wiktionary">ویکی لغت — فارسی سے انگریزی (CC BY-SA 3.0)</string>
|
||||
<string name="source_daneshjoo">دانشجو لغت — فارسی سے انگریزی</string>
|
||||
<string name="dictionary">لغت نامہ</string>
|
||||
<string name="source_wiktionary_ur">ویکی لغت — اردو سے انگریزی (CC BY-SA 3.0)</string>
|
||||
<string name="source_urwiktionary">اردو ویکی لغت — اردو سے اردو (CC BY-SA 3.0)</string>
|
||||
<string name="did_you_mean">ملتے جلتے الفاظ</string>
|
||||
</resources>
|
||||
|
||||
@ -70,4 +70,7 @@
|
||||
<string name="source_wiktionary">Wiktionary — Persian to English (CC BY-SA 3.0)</string>
|
||||
<string name="source_daneshjoo">Daneshjoo — Persian to English</string>
|
||||
<string name="dictionary">Dictionary</string>
|
||||
<string name="source_wiktionary_ur">Wiktionary — Urdu to English (CC BY-SA 3.0)</string>
|
||||
<string name="source_urwiktionary">Urdu Wiktionary — Urdu to Urdu (CC BY-SA 3.0)</string>
|
||||
<string name="did_you_mean">Similar words</string>
|
||||
</resources>
|
||||
|
||||
@ -1,6 +1,7 @@
|
||||
package com.ganjoor.android
|
||||
|
||||
import com.ganjoor.android.data.affixes
|
||||
import com.ganjoor.android.data.letterOverlap
|
||||
import com.ganjoor.android.data.normalise
|
||||
import com.ganjoor.android.data.wordAt
|
||||
import org.junit.Assert.assertEquals
|
||||
@ -106,3 +107,31 @@ class PersianMorphologyTest {
|
||||
assertTrue(affixes("بها").none { it.length < 2 })
|
||||
}
|
||||
}
|
||||
|
||||
class LetterOverlapTest {
|
||||
@Test
|
||||
fun `an identical word overlaps completely`() {
|
||||
assertEquals(1f, letterOverlap("عشق", "عشق"), 0.001f)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `a suffixed form still scores high against its stem`() {
|
||||
assertTrue(letterOverlap("مشکل", "مشکلها") > 0.6f)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `sharing only a first letter scores low`() {
|
||||
assertTrue(letterOverlap("عشق", "عبادتگاه") < 0.4f)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `letters are counted once each, not by presence alone`() {
|
||||
// ااا against ا shares one letter, not three
|
||||
assertEquals(1f / 3f, letterOverlap("ااا", "ا"), 0.001f)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `an empty word never matches`() {
|
||||
assertEquals(0f, letterOverlap("", "عشق"), 0.001f)
|
||||
}
|
||||
}
|
||||
|
||||
@ -19,6 +19,25 @@ bundled rather than merely linked.
|
||||
|
||||
Nothing in Ganjoor's repositories restricts AI-assisted use.
|
||||
|
||||
## Dictionary
|
||||
|
||||
Four sources, each row in the database tagged with the one it came from so the app can name it
|
||||
and so any of them can be dropped without rebuilding the others.
|
||||
|
||||
| Source | Direction | Licence |
|
||||
|---|---|---|
|
||||
| [Wiktionary](https://en.wiktionary.org) | Persian → English | CC BY-SA 3.0 |
|
||||
| [Wiktionary](https://en.wiktionary.org) | Urdu → English | CC BY-SA 3.0 |
|
||||
| [Urdu Wiktionary](https://ur.wiktionary.org) | Urdu → Urdu | CC BY-SA 3.0 |
|
||||
| [Daneshjoo](https://github.com/0xdolan/Daneshjoo) | Persian → English | Repository states MIT |
|
||||
|
||||
Because three of the four are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**.
|
||||
Attribution is shown in the app beside every definition. See [`../tools/README.md`](../tools/README.md)
|
||||
to rebuild it.
|
||||
|
||||
The Daneshjoo entry is worth a caveat: the repository states MIT, but the underlying lexicon is
|
||||
a published Iranian dictionary, so that relicensing is worth verifying before relying on it.
|
||||
|
||||
## Fonts
|
||||
|
||||
| Font | Copyright | Licence | File |
|
||||
|
||||
@ -30,6 +30,20 @@ tables, etymology templates, IPA and descendants. The build keeps the definition
|
||||
form→lemma index (149,589 pairs) and drops the rest. That index is what resolves conjugated
|
||||
verbs: `افتاد → افتادن`, `بگشاید → گشودن`, `دانند → دانستن`.
|
||||
|
||||
## Urdu sources
|
||||
|
||||
```sh
|
||||
curl -L -o ur.jsonl https://kaikki.org/dictionary/Urdu/kaikki.org-dictionary-Urdu.jsonl
|
||||
curl -L -o urwikt.xml.bz2 \
|
||||
https://dumps.wikimedia.org/urwiktionary/latest/urwiktionary-latest-pages-articles.xml.bz2
|
||||
bunzip2 -k urwikt.xml.bz2
|
||||
```
|
||||
|
||||
The first gives Urdu headwords glossed in English. The second is Urdu Wiktionary itself, the only
|
||||
source here whose definitions are written **in Urdu** — thin (around 3,100 usable entries out of
|
||||
31,000 pages, many being stubs), but for a word it does carry an Urdu reader is better served by
|
||||
it than by a translation into English.
|
||||
|
||||
## Licences
|
||||
|
||||
Each row carries its `source`, so attribution stays accurate and either source can be dropped
|
||||
|
||||
@ -1,5 +1,11 @@
|
||||
"""Builds the app's dictionary from Wiktionary (CC BY-SA 3.0) and Daneshjoo (MIT).
|
||||
|
||||
Three sources, each row tagged so the app can say which one answered and in which language:
|
||||
wiktionary-fa Persian headwords, English definitions
|
||||
daneshjoo Persian headwords, English definitions
|
||||
wiktionary-ur Urdu headwords, English definitions
|
||||
|
||||
|
||||
Wiktionary's export is 93 MB of linguistic metadata around 0.94 MB of definitions, so this keeps
|
||||
the definitions and the form->lemma index and discards the rest. The `source` column is what
|
||||
keeps the attribution honest and lets either source be dropped later.
|
||||
@ -36,13 +42,32 @@ for line in open('fa.jsonl', encoding='utf-8'):
|
||||
if gs:
|
||||
pos = e.get('pos') or ''
|
||||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary'))
|
||||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa'))
|
||||
for f in e.get('forms', []):
|
||||
t = f.get('form')
|
||||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||||
forms.add((normalise(t), normalise(word)))
|
||||
|
||||
print(f"wiktionary: {len(entries)} entries, {len(forms)} forms")
|
||||
print(f"wiktionary-fa: {len(entries)} entries, {len(forms)} forms")
|
||||
|
||||
# Urdu shares a great deal of vocabulary with Persian, so these entries answer words the
|
||||
# Persian sources miss. Headwords are Urdu; the definitions are still English.
|
||||
n_fa = len(entries)
|
||||
for line in open('ur.jsonl', encoding='utf-8'):
|
||||
try: e = json.loads(line)
|
||||
except Exception: continue
|
||||
word = e.get('word')
|
||||
if not word: continue
|
||||
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
|
||||
if gs:
|
||||
pos = e.get('pos') or ''
|
||||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur'))
|
||||
for f in e.get('forms', []):
|
||||
t = f.get('form')
|
||||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||||
forms.add((normalise(t), normalise(word)))
|
||||
print(f"wiktionary-ur: {len(entries) - n_fa} entries, {len(forms)} forms total")
|
||||
|
||||
tag = re.compile(r'<[^>]+>')
|
||||
n0 = len(entries)
|
||||
@ -58,6 +83,42 @@ for k, v in MDX('daneshjoo.mdx').items():
|
||||
entries.append((normalise(word), word, txt[:600], 'daneshjoo'))
|
||||
print(f"daneshjoo : {len(entries) - n0} entries")
|
||||
|
||||
# Urdu Wiktionary, the only source here whose definitions are written in Urdu rather than
|
||||
# English. Thin — a few thousand usable entries, many pages being stubs — but for a word it
|
||||
# does carry, an Urdu reader is better served by it than by a translation into English.
|
||||
if os.path.exists('urwikt.xml'):
|
||||
import html as _html
|
||||
raw = open('urwikt.xml', encoding='utf-8', errors='ignore').read()
|
||||
n_ur = len(entries)
|
||||
for title, ns, body in re.findall(
|
||||
r'<title>(.*?)</title>.*?<ns>(\d+)</ns>.*?<text[^>]*>(.*?)</text>', raw, re.S
|
||||
):
|
||||
if ns != '0' or not re.match(r'^[\u0600-\u06FF]', title):
|
||||
continue
|
||||
t = re.sub(r'\{\{[^}]*\}\}', ' ', body)
|
||||
t = re.sub(r'\[\[([^\]|]*\|)?([^\]]*)\]\]', r'\2', t)
|
||||
t = re.sub(r"'{2,}|<[^>]+>", '', _html.unescape(t))
|
||||
section = re.search(r'==\s*معانی\s*==(.*?)(?:\n==|\Z)', t, re.S)
|
||||
lines = (section.group(1) if section else
|
||||
'\n'.join(l.strip(' #') for l in t.split('\n') if l.strip().startswith('#')))
|
||||
kept = []
|
||||
for line in lines.split('\n'):
|
||||
line = re.sub(r'^\d+\.\s*', '', line.strip())
|
||||
# ؎ introduces a verse citation, and "ref"/a year starts the source note; the
|
||||
# definition itself is what comes before either.
|
||||
if line.startswith('؎') or re.match(r'^\(?\s*\d{3,4}ء', line):
|
||||
break
|
||||
line = re.split(r'\bref\b|؎', line)[0].strip()
|
||||
if not line or not re.search(r'[\u0600-\u06FF]', line):
|
||||
continue
|
||||
kept.append(line)
|
||||
if len(' '.join(kept)) > 220:
|
||||
break
|
||||
gloss = ' '.join(kept).strip()[:300]
|
||||
if len(gloss) > 3:
|
||||
entries.append((normalise(title), title, gloss, 'urwiktionary'))
|
||||
print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)")
|
||||
|
||||
c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries)
|
||||
c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms))
|
||||
c.executescript("""
|
||||
|
||||
Loading…
Reference in New Issue
Block a user