Merge pull request 'Dictionary coverage, pronunciation, and chapter order' (#1) from feature/dictionary-coverage-and-chapter-order into main

Reviewed-on: #1
This commit is contained in:
anas 2026-10-04 16:12:01 +00:00
commit 9445f44759
22 changed files with 874 additions and 97 deletions

View File

@ -77,6 +77,15 @@ then with an affix stripped, then the parts of a ZWNJ compound. The lemma step i
classical verse readable — `افتاد` is only findable as `افتادن`, and Wiktionary ships 149,589
form→lemma pairs that make that possible.
Each entry carries its pronunciation where Wiktionary has one — 101,306 of them, tagged with the
variety, Classical Persian first, because a word in a 14th-century ghazal was not said the way
Tehran says it now. Urdu Wiktionary adds the vowelled spelling and the syllable split in Urdu
script. When nothing matches at all, the sheet offers near words ranked by shared letters.
Poems that Ganjoor has recordings for show a play button, streamed rather than stored: a famous
ghazal often has a dozen readings, and downloading them would dwarf the poems. It is the one
part of the app that needs a connection, and it simply doesn't appear without one.
See [`tools/README.md`](tools/README.md) to rebuild it, and for the licensing of each source.
## Where the poems come from

Binary file not shown.

View File

@ -46,7 +46,7 @@ class MainActivity : ComponentActivity() {
val systemInDark = resources.configuration.uiMode and
Configuration.UI_MODE_NIGHT_MASK == Configuration.UI_MODE_NIGHT_YES
window.setBackgroundDrawable(
windowBackground(settings.value.theme, systemInDark).toDrawable()
windowBackground(settings.value.theme, systemInDark, settings.value.oled).toDrawable()
)
enableEdgeToEdge()
@ -62,7 +62,7 @@ class MainActivity : ComponentActivity() {
// navigates right-to-left whichever UI language is selected.
LocalLayoutDirection provides LayoutDirection.Rtl,
) {
GanjoorTheme(settings.value.theme, settings.value.language) {
GanjoorTheme(settings.value.theme, settings.value.language, settings.value.oled) {
GanjoorApp()
}
}

View File

@ -11,13 +11,19 @@ import java.text.Normalizer
/** One definition, and where it came from, so the credit stays attached to the text. */
data class Definition(val word: String, val gloss: String, val source: String)
/**
* How a word sounds. [label] is the variety it belongs to — Classical Persian, Dari, Standard
* Urdu — because a word in a 14th-century ghazal was not said the way Tehran says it now.
*/
data class Pronunciation(val text: String, val label: String)
/**
* Word lookup over a bundled SQLite built from Wiktionary (CC BY-SA 3.0) and Daneshjoo.
* See `tools/build_dictionary.py`; the asset ships gzipped and is unpacked once on first use.
*/
object Dictionary {
/** Guards against a copy interrupted half-way leaving an unopenable file behind. */
private const val ASSET_BYTES = 22_298_624L
private const val ASSET_BYTES = 94384128L
// Not a .gz: the build packager silently gunzips those and drops the extension, which left
// the asset under a different name than the code was opening.
@ -46,6 +52,37 @@ object Dictionary {
}
}
/**
* How [raw] is pronounced, Classical Persian first: this is an app for poetry written long
* before modern Tehrani vowels, and the IPA's dots and stress marks are the syllable
* breakdown that goes with it.
*/
suspend fun pronunciations(raw: String, limit: Int = 5): List<Pronunciation> =
withContext(Dispatchers.IO) {
val database = open() ?: return@withContext emptyList()
val word = normalise(raw).takeIf { it.isNotEmpty() } ?: return@withContext emptyList()
database.rawQuery(
"""
SELECT text, label FROM pron WHERE word = ?
ORDER BY CASE
WHEN label LIKE 'Classical%' THEN 0
WHEN source = 'urwiktionary' THEN 1
WHEN source = 'wiktionary-fa' THEN 2
WHEN source = 'wiktionary-ur' THEN 3
ELSE 4
END
LIMIT ?
""",
arrayOf(word, limit.toString()),
).use { cursor ->
buildList {
while (cursor.moveToNext()) {
add(Pronunciation(cursor.getString(0), cursor.getString(1)))
}
}
}
}
/**
* Looks a word up, widening the search until something matches:
*
@ -69,11 +106,42 @@ object Dictionary {
.firstNotNullOfOrNull { direct(database, normalise(it)).ifEmpty { null } }
.orEmpty()
}
// A selection is usually a phrase rather than a word; fall back to its words.
.ifEmpty {
raw.split(' ', '\n', '\r')
.map { normalise(it) }
.filter { it.length > 1 && it != word }
.firstNotNullOfOrNull { part ->
direct(database, part).ifEmpty {
lemmas(database, part).flatMap { direct(database, it) }.ifEmpty { null }
}
}
.orEmpty()
}
}
/**
* Ordered the way a reader of this app wants to be answered: a definition written in Urdu
* first, because it needs no translating at all, then the Persian sources, then the ones
* keyed on another language. English is what the rest fall back to, so it comes last by
* coming from the sources that sit last.
*
* Arabic is last outright: its forms index is larger than every other source combined,
* which makes it the likeliest to match something by coincidence.
*/
private fun direct(database: SQLiteDatabase, word: String): List<Definition> =
database.rawQuery(
"SELECT display, gloss, source FROM entry WHERE word = ? LIMIT 12",
"""
SELECT display, gloss, source FROM entry WHERE word = ?
ORDER BY CASE source
WHEN 'urwiktionary' THEN 0
WHEN 'wiktionary-fa' THEN 1
WHEN 'daneshjoo' THEN 2
WHEN 'wiktionary-ur' THEN 3
ELSE 4
END
LIMIT 12
""",
arrayOf(word),
).use { cursor ->
buildList {
@ -83,6 +151,47 @@ object Dictionary {
}
}
/**
* Headwords that look like [raw], for when nothing matched exactly — a misread letter, an
* unusual spelling, or a word the dictionary simply spells differently.
*
* Candidates come from an index range scan on the first letters, then are ranked by how
* many letters they share with the query. ponytail: shared letters rather than an edit
* distance, which would need the whole table scanned to be worth the extra precision.
*/
suspend fun suggest(raw: String, limit: Int = 6): List<String> = withContext(Dispatchers.IO) {
val database = open() ?: return@withContext emptyList()
val word = normalise(raw).takeIf { it.length > 1 } ?: return@withContext emptyList()
// Widen the prefix until there is something to rank, but never scan the whole table.
val candidates = generateSequence(minOf(3, word.length - 1)) { (it - 1).takeIf { n -> n >= 1 } }
.map { prefixLength -> byPrefix(database, word.take(prefixLength)) }
.firstOrNull { it.size >= 3 }
?: return@withContext emptyList()
candidates
.asSequence()
.filter { it.first != word }
.map { (normalised, display) -> display to letterOverlap(word, normalised) }
.filter { it.second > 0.45f }
.sortedByDescending { it.second }
.map { it.first }
.distinct()
.take(limit)
.toList()
}
/** Index range scan: everything whose normalised form starts with [prefix]. */
private fun byPrefix(database: SQLiteDatabase, prefix: String): List<Pair<String, String>> =
database.rawQuery(
"SELECT DISTINCT word, display FROM entry WHERE word >= ? AND word < ? LIMIT 400",
arrayOf(prefix, prefix + '\uFFFF'),
).use { cursor ->
buildList {
while (cursor.moveToNext()) add(cursor.getString(0) to cursor.getString(1))
}
}
private fun lemmas(database: SQLiteDatabase, form: String): List<String> =
database.rawQuery(
"SELECT lemma FROM form WHERE form = ? LIMIT 6",
@ -94,14 +203,29 @@ object Dictionary {
private const val ZWNJ = '‌'
private val SUFFIXES = listOf("ها", "اش", "ش", "م", "ت", "را", "ی", "ان")
// "ال" is the Arabic definite article: poems quote Arabic, so السّاقی has to reach ساقی.
private val PREFIXES = listOf("ال", "می", "بر", "ب")
/**
* Longest first, so تربتش strips شـ rather than matching nothing. These are the endings that
* actually turn up in classical verse: plurals, the object marker, and the enclitic pronouns
* that Persian glues onto a verb or noun — آیدت is آید + ت, باشدش is باشد + ش.
*/
private val SUFFIXES = listOf(
"شان", "تان", "مان", "ها", "اش", "ست", "یم", "ید", "ند", "را", "ش", "م", "ت", "ی", "ان", "ه",
)
/** Candidate stems after stripping one common affix. Order matters: longest affix first. */
internal fun affixes(word: String): List<String> = buildList {
SUFFIXES.forEach { if (word.endsWith(it) && word.length > it.length + 1) add(word.dropLast(it.length)) }
PREFIXES.forEach { if (word.startsWith(it) && word.length > it.length + 1) add(word.drop(it.length)) }
/** "ال" is the Arabic definite article, "ن"/"نمی" negation, "بی" privative. */
private val PREFIXES = listOf("نمی", "ال", "می", "بی", "بر", "ن", "ب")
/**
* Candidate stems, one affix deep and then two — برنیاید is بر + ن + یاید, and a single pass
* would never reach the verb. Ordered so the least mangled candidate is tried first.
*/
internal fun affixes(word: String): List<String> {
fun oneStep(w: String) = buildList {
SUFFIXES.forEach { if (w.endsWith(it) && w.length > it.length + 1) add(w.dropLast(it.length)) }
PREFIXES.forEach { if (w.startsWith(it) && w.length > it.length + 1) add(w.drop(it.length)) }
}
val first = oneStep(word)
return (first + first.flatMap(::oneStep)).distinct()
}
private val HARAKAT = (0x064B..0x0652) + listOf(0x0670, 0x0640) + (0x0610..0x0615)
@ -125,6 +249,17 @@ internal fun normalise(text: String, keepZwnj: Boolean = false): String {
return (if (keepZwnj) folded else folded.replace(ZWNJ.toString(), "")).trim()
}
/**
* How much of two words' letters coincide — the overlapping letters counted against the longer
* word, so مشکل and مشکلها score high while a word that merely starts the same does not.
*/
internal fun letterOverlap(a: String, b: String): Float {
if (a.isEmpty() || b.isEmpty()) return 0f
val remaining = b.toMutableList()
val shared = a.count { remaining.remove(it) }
return shared.toFloat() / maxOf(a.length, b.length)
}
/** The whole word surrounding [index], for turning a tap into something to look up. */
internal fun wordAt(text: String, index: Int): String? {
if (text.isEmpty()) return null

View File

@ -145,6 +145,18 @@ private data class LiveCat(val poems: List<LivePoem> = emptyList())
@Serializable
private data class LivePoem(val id: Int = 0, val excerpt: String? = null)
/** A reading of a whole poem, hosted by Ganjoor. */
@Serializable
data class Recitation(
val id: Int = 0,
val audioTitle: String = "",
val audioArtist: String = "",
val mp3Url: String = "",
)
@Serializable
private data class LivePoemRecitations(val recitations: List<Recitation> = emptyList())
private val liveJson = Json { ignoreUnknownKeys = true }
@Serializable
@ -166,6 +178,37 @@ fun catPath(fullUrl: String) = "poets${fullUrl.trimEnd('/')}/_cat.json"
fun poemPath(fullUrl: String) = "poets${fullUrl.trimEnd('/')}.json"
/** A row in a category listing: either a chapter to open, or a poem to read. */
sealed interface CatEntry {
data class Chapter(val category: Category) : CatEntry
data class Poem(val poem: PoemRef) : CatEntry
}
private val PREFACE_TITLES = listOf("دیباچه", "مقدمه", "سرآغاز", "پیشگفتار", "آغاز")
/**
* Orders a category the way ganjoor.net does: a book's own preface first, then its chapters,
* then whatever other poems sit directly under it. Golestan's دیباچه belongs above the eight
* باب, not below them; Hafez's مقدّمه above his five collections, with مثنوی and ساقی‌نامه after.
*
* Ganjoor decides this with each poem's MixedModeOrder — 1 sorts a poem above the chapters, 0
* below — but that field isn't in the exported `_cat.json`, only on the live API's per-poem
* record, which would be one request per poem.
*
* ponytail: so prefaces are recognised by title instead. Adding MixedModeOrder to the Poems
* entries in ganjoor-data would make this exact; until then a book whose preface is named
* something unusual still lands after its chapters.
*/
fun orderedEntries(category: Category): List<CatEntry> {
val (prefaces, rest) = category.poems.partition { poem ->
val title = normalise(poem.title).trimStart()
PREFACE_TITLES.any { title.startsWith(it) }
}
return prefaces.map(CatEntry::Poem) +
category.childCats.map(CatEntry::Chapter) +
rest.map(CatEntry::Poem)
}
/** One step of a poem's path. [url] is null for the poem itself, which is already open. */
data class Crumb(val label: String, val url: String?)
@ -324,6 +367,27 @@ object Ganjoor {
}
}
/**
* Readings of a poem, by the people who recorded them for ganjoor.net.
*
* Streamed, never stored: the files are a few hundred kilobytes each and there are often a
* dozen readings of a famous ghazal, so downloading them all would dwarf the poems. Offline
* mode therefore has none of this, which is honest — a recording is the one thing here that
* genuinely needs the network.
*/
suspend fun recitations(poemId: Int): List<Recitation> = withContext(Dispatchers.IO) {
if (offline || poemId == 0) return@withContext emptyList()
val url = liveBase.newBuilder()
.addPathSegments("api/ganjoor/poem/$poemId")
.addQueryParameter("recitations", "true")
.addQueryParameter("verseDetails", "false")
.build()
runCatching {
liveJson.decodeFromString<LivePoemRecitations>(fetch(url)).recitations
.filter { it.mp3Url.isNotBlank() }
}.getOrDefault(emptyList())
}
/**
* Saves a poet's whole tree — biography, every category and every poem — for offline reading.
* Re-running it is cheap: anything already on disk is skipped, which is also how a download

View File

@ -64,11 +64,17 @@ private val CREDITS = listOf(
),
Credit(
"Wiktionary",
"Wiktionary contributors — the word definitions, and the inflected-form index that " +
"finds a conjugated verb's dictionary entry",
"Wiktionary contributors — Persian and Urdu definitions, and the inflected-form " +
"index that finds a conjugated verb's dictionary entry",
"CC BY-SA 3.0. The bundled dictionary is therefore also CC BY-SA 3.0.",
url = "https://en.wiktionary.org",
),
Credit(
"Urdu Wiktionary",
"Urdu Wiktionary contributors — the definitions written in Urdu rather than English",
"CC BY-SA 3.0. The bundled dictionary is therefore also CC BY-SA 3.0.",
url = "https://ur.wiktionary.org",
),
Credit(
"Daneshjoo Dictionary",
"Layered under Wiktionary for the words it doesn't carry",

View File

@ -38,9 +38,11 @@ import androidx.compose.ui.res.stringResource
import androidx.compose.ui.text.style.TextOverflow
import androidx.compose.ui.unit.dp
import com.ganjoor.android.R
import com.ganjoor.android.data.CatEntry
import com.ganjoor.android.data.Downloads
import com.ganjoor.android.data.Ganjoor
import com.ganjoor.android.data.Offline
import com.ganjoor.android.data.orderedEntries
@OptIn(ExperimentalMaterial3Api::class)
@Composable
@ -63,6 +65,8 @@ fun CategoryScreen(
}
}
val entries = remember(cat) { orderedEntries(cat) }
Scaffold(
topBar = {
TopAppBar(
@ -94,12 +98,25 @@ fun CategoryScreen(
cat.description?.takeIf { it.isNotBlank() }?.let { description ->
item { Description(description) }
}
items(cat.childCats, key = { "c${it.id}" }) { child ->
NavRow(child.title, isCategory = true) { onCategory(child.fullUrl) }
}
items(cat.poems, key = { "p${it.id}" }) { poem ->
NavRow(poem.title, isCategory = false, excerpt = excerpts[poem.id]) {
onPoem(poem.fullUrl)
items(
items = entries,
key = { entry ->
when (entry) {
is CatEntry.Chapter -> "c${entry.category.id}"
is CatEntry.Poem -> "p${entry.poem.id}"
}
},
) { entry ->
when (entry) {
is CatEntry.Chapter -> NavRow(entry.category.title, isCategory = true) {
onCategory(entry.category.fullUrl)
}
is CatEntry.Poem -> NavRow(
title = entry.poem.title,
isCategory = false,
excerpt = excerpts[entry.poem.id],
) { onPoem(entry.poem.fullUrl) }
}
}
}

View File

@ -105,8 +105,14 @@ fun PoemScreen(
)
}
) { insets ->
// Free-form selection for copying any span; the per-couplet actions below are for
// saving a passage with the reference attached, which a raw copy would lose.
// Free-form selection for copying any span; tapping a word looks it up, and the
// per-couplet actions save a passage with the reference attached, which a raw copy
// would lose.
//
// ponytail: no dictionary entry in the selection toolbar. Compose 1.10 stopped
// routing SelectionContainer through LocalTextToolbar — a custom TextToolbar is
// simply never asked to show — and the replacement, foundation's contextmenu
// package, is internal. Revisit when that becomes public API.
SelectionContainer {
LazyColumn(
modifier = Modifier.fillMaxSize(),
@ -120,6 +126,7 @@ fun PoemScreen(
item {
Column(Modifier.padding(bottom = 12.dp)) {
Breadcrumbs(poem.fullTitle, poem.fullUrl.ifBlank { fullUrl }, onCategory)
RecitationPlayer(poem.id)
poem.metre?.rhythm?.let { rhythm ->
Text(
text = rhythm,

View File

@ -78,6 +78,15 @@ fun ReadingSettingsSheet(onDismiss: () -> Unit) {
settings.update { it.copy(theme = mode) }
}
// Only means anything on a dark theme, so it sits with them and says so.
Toggle(
title = R.string.oled,
note = R.string.oled_note,
checked = prefs.oled,
) { on ->
settings.update { it.copy(oled = on) }
}
Label(R.string.font)
Chips(ReadingFont.entries, prefs.font, { stringResource(it.label) }) { font ->
settings.update { it.copy(font = font) }

View File

@ -0,0 +1,141 @@
package com.ganjoor.android.ui
import android.media.AudioAttributes
import android.media.MediaPlayer
import androidx.compose.foundation.layout.Arrangement
import androidx.compose.foundation.layout.Row
import androidx.compose.foundation.layout.fillMaxWidth
import androidx.compose.foundation.layout.padding
import androidx.compose.foundation.layout.size
import androidx.compose.material.icons.Icons
import androidx.compose.material.icons.filled.PlayArrow
import androidx.compose.material3.CircularProgressIndicator
import androidx.compose.material3.DropdownMenu
import androidx.compose.material3.DropdownMenuItem
import androidx.compose.material3.Icon
import androidx.compose.material3.IconButton
import androidx.compose.material3.MaterialTheme
import androidx.compose.material3.Text
import androidx.compose.material3.TextButton
import androidx.compose.runtime.Composable
import androidx.compose.runtime.DisposableEffect
import androidx.compose.runtime.getValue
import androidx.compose.runtime.mutableIntStateOf
import androidx.compose.runtime.mutableStateOf
import androidx.compose.runtime.produceState
import androidx.compose.runtime.remember
import androidx.compose.runtime.setValue
import androidx.compose.ui.Alignment
import androidx.compose.ui.Modifier
import androidx.compose.ui.res.painterResource
import androidx.compose.ui.res.stringResource
import androidx.compose.ui.text.style.TextOverflow
import androidx.compose.ui.unit.dp
import com.ganjoor.android.R
import com.ganjoor.android.data.Ganjoor
import com.ganjoor.android.data.Recitation
/**
* Plays a reading of the poem, streamed from Ganjoor.
*
* ponytail: the platform's MediaPlayer rather than ExoPlayer — one URL, play and pause, no
* playlist or seeking to justify a media library. Nothing is cached, so this is the one part of
* the app that needs a connection; it simply doesn't appear when there is no reading or no
* network.
*/
@Composable
fun RecitationPlayer(poemId: Int) {
val recitations by produceState(emptyList<Recitation>(), poemId) {
value = Ganjoor.recitations(poemId)
}
if (recitations.isEmpty()) return
var chosen by remember(poemId) { mutableIntStateOf(0) }
var playing by remember(poemId) { mutableStateOf(false) }
var loading by remember(poemId) { mutableStateOf(false) }
var picking by remember { mutableStateOf(false) }
val player = remember {
MediaPlayer().apply {
setAudioAttributes(
AudioAttributes.Builder()
.setUsage(AudioAttributes.USAGE_MEDIA)
.setContentType(AudioAttributes.CONTENT_TYPE_MUSIC)
.build()
)
}
}
// A reading left playing when the screen goes would keep the whole poem in memory.
DisposableEffect(player) { onDispose { runCatching { player.release() } } }
val recitation = recitations.getOrNull(chosen) ?: return
// Changing reader stops whatever was playing, so the two never overlap.
DisposableEffect(recitation.mp3Url) {
runCatching { player.reset() }
playing = false
loading = false
onDispose { }
}
fun toggle() {
if (playing) {
runCatching { player.pause() }
playing = false
return
}
if (player.currentPosition > 0) {
runCatching { player.start() }.onSuccess { playing = true }
return
}
loading = true
runCatching {
player.reset()
player.setDataSource(recitation.mp3Url)
player.setOnPreparedListener { it.start(); playing = true; loading = false }
player.setOnCompletionListener { playing = false }
player.setOnErrorListener { _, _, _ -> playing = false; loading = false; true }
player.prepareAsync()
}.onFailure { loading = false }
}
Row(
modifier = Modifier.fillMaxWidth().padding(bottom = 8.dp),
verticalAlignment = Alignment.CenterVertically,
horizontalArrangement = Arrangement.spacedBy(4.dp),
) {
IconButton(onClick = ::toggle) {
when {
loading -> CircularProgressIndicator(Modifier.size(20.dp), strokeWidth = 2.dp)
// Core Material icons ship no pause glyph, and the extended set is 4 MB for one.
playing -> Icon(painterResource(R.drawable.ic_pause), stringResource(R.string.pause))
else -> Icon(Icons.Default.PlayArrow, stringResource(R.string.play_recitation))
}
}
TextButton(
onClick = { if (recitations.size > 1) picking = true },
modifier = Modifier.weight(1f, fill = false),
) {
Text(
text = recitation.audioArtist.ifBlank { stringResource(R.string.play_recitation) },
style = MaterialTheme.typography.labelLarge,
maxLines = 1,
overflow = TextOverflow.Ellipsis,
)
}
if (recitations.size > 1) {
Text(
text = "${chosen + 1}/${recitations.size}",
style = MaterialTheme.typography.labelSmall,
color = MaterialTheme.colorScheme.onSurfaceVariant,
)
}
DropdownMenu(expanded = picking, onDismissRequest = { picking = false }) {
recitations.forEachIndexed { index, item ->
DropdownMenuItem(
text = { Text(item.audioArtist.ifBlank { item.audioTitle }) },
onClick = { chosen = index; picking = false },
)
}
}
}
}

View File

@ -18,9 +18,6 @@ enum class ThemeMode(@StringRes val label: Int) {
Dark(R.string.theme_dark),
Sepia(R.string.theme_sepia),
SepiaDark(R.string.theme_sepia_dark),
/** Pure black, so OLED panels can switch the pixels off entirely. */
Black(R.string.theme_black),
}
enum class ReadingFont(@StringRes val label: Int) {
@ -58,6 +55,8 @@ data class Prefs(
val showSummaries: Boolean = false,
val language: Language = Language.Fa,
val offline: Boolean = false,
/** True black backgrounds, applied to whichever dark theme is in use. */
val oled: Boolean = false,
val poetSort: PoetSort = PoetSort.Default,
)
@ -74,6 +73,8 @@ class Settings(context: Context) {
showSummaries = prefs.getBoolean("showSummaries", false),
language = languageOrDefault(prefs.getString(KEY_LANGUAGE, null)),
offline = prefs.getBoolean("offline", false),
// "Black" used to be a sixth theme; it is a flag on the dark ones now.
oled = prefs.getBoolean("oled", prefs.getString("theme", null) == "Black"),
poetSort = enumOrDefault(prefs.getString("poetSort", null), PoetSort.Default),
)
)
@ -91,6 +92,7 @@ class Settings(context: Context) {
putBoolean("showSummaries", p.showSummaries)
putString(KEY_LANGUAGE, p.language.tag)
putBoolean("offline", p.offline)
putBoolean("oled", p.oled)
putString("poetSort", p.poetSort.name)
}
}

View File

@ -2,22 +2,28 @@ package com.ganjoor.android.ui
import androidx.compose.foundation.layout.Arrangement
import androidx.compose.foundation.layout.Column
import androidx.compose.foundation.layout.Row
import androidx.compose.foundation.layout.fillMaxWidth
import androidx.compose.foundation.layout.navigationBarsPadding
import androidx.compose.foundation.layout.padding
import androidx.compose.foundation.rememberScrollState
import androidx.compose.foundation.verticalScroll
import androidx.compose.foundation.layout.FlowRow
import androidx.compose.material3.CircularProgressIndicator
import androidx.compose.material3.ExperimentalMaterial3Api
import androidx.compose.material3.HorizontalDivider
import androidx.compose.material3.MaterialTheme
import androidx.compose.material3.ModalBottomSheet
import androidx.compose.material3.SuggestionChip
import androidx.compose.material3.Text
import androidx.compose.runtime.Composable
import androidx.compose.runtime.CompositionLocalProvider
import androidx.compose.runtime.getValue
import androidx.compose.runtime.mutableStateOf
import androidx.compose.runtime.produceState
import androidx.compose.runtime.remember
import androidx.compose.runtime.setValue
import androidx.compose.ui.Alignment
import androidx.compose.ui.Modifier
import androidx.compose.ui.platform.LocalLayoutDirection
import androidx.compose.ui.res.stringResource
@ -26,6 +32,7 @@ import androidx.compose.ui.unit.dp
import com.ganjoor.android.R
import com.ganjoor.android.data.Definition
import com.ganjoor.android.data.Dictionary
import com.ganjoor.android.data.Pronunciation
import com.ganjoor.android.ui.theme.readingStyle
/** English prose inside an otherwise right-to-left sheet. */
@ -34,12 +41,43 @@ private fun LeftToRight(content: @Composable () -> Unit) {
CompositionLocalProvider(LocalLayoutDirection provides LayoutDirection.Ltr, content = content)
}
/** Lays a definition out the way its own script reads. */
@Composable
private fun InDirectionOf(text: String, content: @Composable () -> Unit) {
val arabicScript = text.count { it in '\u0600'..'\u06FF' }
val latin = text.count { it in 'A'..'Z' || it in 'a'..'z' }
CompositionLocalProvider(
LocalLayoutDirection provides
if (arabicScript > latin) LayoutDirection.Rtl else LayoutDirection.Ltr,
content = content,
)
}
/** Which dictionary answered, and in which language pair. */
private fun sourceLabel(source: String) = when (source) {
"wiktionary-fa" -> R.string.source_wiktionary
"wiktionary-ur" -> R.string.source_wiktionary_ur
"urwiktionary" -> R.string.source_urwiktionary
"wiktionary-ar" -> R.string.source_wiktionary_ar
else -> R.string.source_daneshjoo
}
/** What the dictionary knows about a tapped word. */
@OptIn(ExperimentalMaterial3Api::class)
@Composable
fun WordSheet(word: String, onDismiss: () -> Unit) {
val prefs = LocalSettings.current.value
val definitions by produceState<List<Definition>?>(null, word) { value = Dictionary.lookup(word) }
// A suggestion replaces what is being looked up, so the sheet can be followed like a trail.
var current by remember(word) { mutableStateOf(word) }
val definitions by produceState<List<Definition>?>(null, current) {
value = Dictionary.lookup(current)
}
val sounds by produceState(emptyList<Pronunciation>(), current) {
value = Dictionary.pronunciations(current)
}
val suggestions by produceState(emptyList<String>(), current, definitions) {
value = if (definitions?.isEmpty() == true) Dictionary.suggest(current) else emptyList()
}
ModalBottomSheet(onDismissRequest = onDismiss) {
Column(
@ -52,47 +90,90 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
) {
// The headword in the reading font, at reading size: it is a line of poetry, after all.
Text(
text = word,
text = current,
style = readingStyle(prefs.font, prefs.fontSize, prefs.fontWeight.weight),
modifier = Modifier.fillMaxWidth(),
)
if (sounds.isNotEmpty()) {
// One direction for the whole block, and one pronunciation per line. Flowing
// them side by side put Latin IPA and Urdu spelling in the same right-to-left
// run, which reordered the chips and left each label under someone else's value.
LeftToRight {
Column(verticalArrangement = Arrangement.spacedBy(2.dp)) {
sounds.forEach { sound ->
Row(
modifier = Modifier.fillMaxWidth(),
horizontalArrangement = Arrangement.spacedBy(10.dp),
verticalAlignment = Alignment.CenterVertically,
) {
Text(
text = sound.text,
style = MaterialTheme.typography.bodyMedium,
modifier = Modifier.weight(1f, fill = false),
)
if (sound.label.isNotBlank()) {
Text(
text = sound.label,
style = MaterialTheme.typography.labelSmall,
color = MaterialTheme.colorScheme.onSurfaceVariant,
)
}
}
}
}
}
}
HorizontalDivider()
when {
definitions == null -> CircularProgressIndicator(Modifier.padding(vertical = 16.dp))
definitions!!.isEmpty() -> LeftToRight {
Text(
text = stringResource(R.string.no_definition),
style = MaterialTheme.typography.bodyMedium,
color = MaterialTheme.colorScheme.onSurfaceVariant,
modifier = Modifier.fillMaxWidth(),
)
definitions!!.isEmpty() -> Column(
verticalArrangement = Arrangement.spacedBy(8.dp),
) {
LeftToRight {
Text(
text = stringResource(R.string.no_definition),
style = MaterialTheme.typography.bodyMedium,
color = MaterialTheme.colorScheme.onSurfaceVariant,
modifier = Modifier.fillMaxWidth(),
)
}
if (suggestions.isNotEmpty()) {
Text(
text = stringResource(R.string.did_you_mean),
style = MaterialTheme.typography.titleSmall,
color = MaterialTheme.colorScheme.primary,
)
FlowRow(horizontalArrangement = Arrangement.spacedBy(8.dp)) {
suggestions.forEach { suggestion ->
SuggestionChip(
onClick = { current = suggestion },
label = { Text(suggestion) },
)
}
}
}
}
else -> definitions!!.forEach { definition ->
Column(Modifier.fillMaxWidth()) {
// The headword actually matched, which may be the lemma rather than the
// word as it appears in the line. Persian, so it stays right-to-left.
if (definition.word != word) {
if (definition.word != current) {
Text(
text = definition.word,
style = MaterialTheme.typography.titleSmall,
color = MaterialTheme.colorScheme.primary,
)
}
// The definitions are English; right-aligning them reads badly.
LeftToRight {
// Most definitions are English and right-aligning them reads badly;
// the Urdu ones are right-to-left like the rest of the app.
InDirectionOf(definition.gloss) {
Column(Modifier.fillMaxWidth()) {
Text(definition.gloss, style = MaterialTheme.typography.bodyMedium)
Text(
text = stringResource(
if (definition.source == "wiktionary") {
R.string.source_wiktionary
} else {
R.string.source_daneshjoo
}
),
text = stringResource(sourceLabel(definition.source)),
style = MaterialTheme.typography.labelSmall,
color = MaterialTheme.colorScheme.onSurfaceVariant,
)

View File

@ -2,12 +2,14 @@ package com.ganjoor.android.ui.theme
import android.app.Activity
import androidx.compose.foundation.isSystemInDarkTheme
import androidx.compose.material3.ColorScheme
import androidx.compose.material3.MaterialTheme
import androidx.compose.material3.darkColorScheme
import androidx.compose.material3.lightColorScheme
import androidx.compose.runtime.Composable
import androidx.compose.runtime.SideEffect
import androidx.compose.ui.graphics.Color
import androidx.compose.ui.graphics.lerp
import androidx.compose.ui.graphics.toArgb
import androidx.compose.ui.platform.LocalView
import androidx.core.graphics.drawable.toDrawable
@ -66,36 +68,6 @@ private val DarkScheme = darkColorScheme(
outline = Color(0xFF899393),
)
/**
* True black, for OLED panels: a black pixel is an unlit pixel, so a night-time reading session
* costs noticeably less battery than the regular dark theme's dark grey. Surfaces step up in
* near-black greys so cards and sheets stay distinguishable without lighting the whole screen.
*/
private val BlackScheme = darkColorScheme(
primary = Color(0xFF80D4DA),
onPrimary = Color(0xFF00363A),
primaryContainer = Color(0xFF004F53),
onPrimaryContainer = Color(0xFF9CF1F6),
secondary = Color(0xFFFFB873),
onSecondary = Color(0xFF4A2800),
secondaryContainer = Color(0xFF693C00),
onSecondaryContainer = Color(0xFFFFDCBE),
background = Color(0xFF000000),
onBackground = Color(0xFFE3E3E3),
surface = Color(0xFF000000),
onSurface = Color(0xFFE3E3E3),
surfaceVariant = Color(0xFF1C1C1C),
onSurfaceVariant = Color(0xFFBDBDBD),
outline = Color(0xFF6E6E6E),
surfaceBright = Color(0xFF262626),
surfaceDim = Color(0xFF000000),
surfaceContainerLowest = Color(0xFF000000),
surfaceContainerLow = Color(0xFF0A0A0A),
surfaceContainer = Color(0xFF101010),
surfaceContainerHigh = Color(0xFF1A1A1A),
surfaceContainerHighest = Color(0xFF242424),
)
// Aged paper, for long reading sessions.
private val SepiaScheme = lightColorScheme(
primary = Color(0xFF7A4E24),
@ -153,28 +125,61 @@ private val SepiaDarkScheme = darkColorScheme(
* so the gap between the window appearing and the first frame matches the theme instead of
* flashing the platform's white. A static XML theme can't express sepia or OLED, hence this.
*/
fun windowBackground(mode: ThemeMode, systemInDark: Boolean): Int = when (mode) {
ThemeMode.Light -> LightScheme
ThemeMode.Dark -> DarkScheme
ThemeMode.Sepia -> SepiaScheme
ThemeMode.SepiaDark -> SepiaDarkScheme
ThemeMode.Black -> BlackScheme
ThemeMode.System -> if (systemInDark) DarkScheme else LightScheme
}.background.toArgb()
fun windowBackground(mode: ThemeMode, systemInDark: Boolean, oled: Boolean): Int {
val dark = when (mode) {
ThemeMode.System -> systemInDark
ThemeMode.Light, ThemeMode.Sepia -> false
ThemeMode.Dark, ThemeMode.SepiaDark -> true
}
if (oled && dark) return Color.Black.toArgb()
return when (mode) {
ThemeMode.Light -> LightScheme
ThemeMode.Dark -> DarkScheme
ThemeMode.Sepia -> SepiaScheme
ThemeMode.SepiaDark -> SepiaDarkScheme
ThemeMode.System -> if (systemInDark) DarkScheme else LightScheme
}.background.toArgb()
}
/**
* Pushes a dark scheme to true black for OLED panels, where an unlit pixel costs no power.
*
* Only the surfaces move, and they move towards black rather than being replaced by it, so sepia
* night keeps its warmth instead of turning into the grey dark theme. Text and accents are left
* exactly as they were.
*/
private fun ColorScheme.asOled(): ColorScheme = copy(
background = Color.Black,
surface = Color.Black,
surfaceDim = Color.Black,
surfaceContainerLowest = Color.Black,
surfaceContainerLow = lerp(surfaceContainerLow, Color.Black, 0.80f),
surfaceContainer = lerp(surfaceContainer, Color.Black, 0.74f),
surfaceContainerHigh = lerp(surfaceContainerHigh, Color.Black, 0.64f),
surfaceContainerHighest = lerp(surfaceContainerHighest, Color.Black, 0.54f),
surfaceBright = lerp(surfaceBright, Color.Black, 0.46f),
surfaceVariant = lerp(surfaceVariant, Color.Black, 0.52f),
)
@Composable
fun GanjoorTheme(mode: ThemeMode, language: Language, content: @Composable () -> Unit) {
fun GanjoorTheme(
mode: ThemeMode,
language: Language,
oled: Boolean,
content: @Composable () -> Unit,
) {
val dark = when (mode) {
ThemeMode.System -> isSystemInDarkTheme()
ThemeMode.Light, ThemeMode.Sepia -> false
ThemeMode.Dark, ThemeMode.SepiaDark, ThemeMode.Black -> true
ThemeMode.Dark, ThemeMode.SepiaDark -> true
}
val scheme = when (mode) {
val chosen = when (mode) {
ThemeMode.Sepia -> SepiaScheme
ThemeMode.SepiaDark -> SepiaDarkScheme
ThemeMode.Black -> BlackScheme
else -> if (dark) DarkScheme else LightScheme
}
// Only dark schemes have anything to gain from it.
val scheme = if (oled && dark) chosen.asOled() else chosen
val view = LocalView.current
if (!view.isInEditMode) {

View File

@ -0,0 +1,9 @@
<vector xmlns:android="http://schemas.android.com/apk/res/android"
android:width="24dp"
android:height="24dp"
android:viewportWidth="24"
android:viewportHeight="24">
<path
android:fillColor="@android:color/white"
android:pathData="M6,19h4V5H6v14zM14,5v14h4V5h-4z" />
</vector>

View File

@ -58,11 +58,18 @@
<string name="search_needs_connection">جست‌وجو به اینترنت نیاز دارد. حالت برون‌خط را خاموش کنید.</string>
<string name="load_more">بیشتر</string>
<string name="done">تمام</string>
<string name="theme_black">سیاه (OLED)</string>
<string name="about">درباره و پروانه‌ها</string>
<string name="about_intro">گنجور برای اندروید نرم‌افزار آزاد است. شعرها، قلم‌ها و همهٔ کتابخانه‌هایی که این برنامه بر آن‌ها ساخته شده در زیر آمده‌اند؛ برای خواندن متن کامل پروانه روی هر مورد بزنید.</string>
<string name="no_definition">برای این واژه مدخلی یافت نشد.</string>
<string name="source_wiktionary">ویکی‌واژه (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">فرهنگ دانشجو</string>
<string name="source_wiktionary">ویکی‌واژه — فارسی به انگلیسی (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">فرهنگ دانشجو — فارسی به انگلیسی</string>
<string name="dictionary">لغت‌نامه</string>
<string name="source_wiktionary_ur">ویکی‌واژه — اردو به انگلیسی (CC BY-SA 3.0)</string>
<string name="source_urwiktionary">ویکی‌واژهٔ اردو — اردو به اردو (CC BY-SA 3.0)</string>
<string name="did_you_mean">واژه‌های نزدیک</string>
<string name="source_wiktionary_ar">ویکی‌واژه — عربی به انگلیسی (CC BY-SA 3.0)</string>
<string name="oled">پس‌زمینهٔ سیاه (OLED)</string>
<string name="oled_note">روی پوسته‌های تاریک اعمال می‌شود و در نمایشگر OLED باتری کمتری می‌برد</string>
<string name="play_recitation">پخش خوانش</string>
<string name="pause">توقف</string>
</resources>

View File

@ -58,11 +58,18 @@
<string name="search_needs_connection">تلاش کے لیے انٹرنیٹ درکار ہے۔ آف لائن موڈ بند کریں۔</string>
<string name="load_more">مزید</string>
<string name="done">مکمل</string>
<string name="theme_black">سیاہ (OLED)</string>
<string name="about">تعارف اور لائسنس</string>
<string name="about_intro">گنجور فار اینڈرائیڈ آزاد سافٹ ویئر ہے۔ کلام، فونٹس اور تمام لائبریریاں ذیل میں درج ہیں؛ مکمل لائسنس پڑھنے کے لیے کسی اندراج پر ٹیپ کریں۔</string>
<string name="no_definition">اس لفظ کا کوئی اندراج نہیں ملا۔</string>
<string name="source_wiktionary">ویکی لغت (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">دانشجو لغت</string>
<string name="source_wiktionary">ویکی لغت — فارسی سے انگریزی (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">دانشجو لغت — فارسی سے انگریزی</string>
<string name="dictionary">لغت نامہ</string>
<string name="source_wiktionary_ur">ویکی لغت — اردو سے انگریزی (CC BY-SA 3.0)</string>
<string name="source_urwiktionary">اردو ویکی لغت — اردو سے اردو (CC BY-SA 3.0)</string>
<string name="did_you_mean">ملتے جلتے الفاظ</string>
<string name="source_wiktionary_ar">ویکی لغت — عربی سے انگریزی (CC BY-SA 3.0)</string>
<string name="oled">سیاہ پس منظر (OLED)</string>
<string name="oled_note">تاریک تھیمز پر لاگو ہوتا ہے؛ OLED اسکرین پر بیٹری بچاتا ہے</string>
<string name="play_recitation">قرات سنیں</string>
<string name="pause">وقفہ</string>
</resources>

View File

@ -63,11 +63,18 @@
<string name="search_needs_connection">Search needs a connection. Turn off offline mode to use it.</string>
<string name="load_more">Load more</string>
<string name="done">Done</string>
<string name="theme_black">Black (OLED)</string>
<string name="about">About &amp; licences</string>
<string name="about_intro">Ganjoor for Android is free software. The poems, the fonts and every library it is built on are credited below; tap an entry to read its full licence.</string>
<string name="no_definition">No entry for this word.</string>
<string name="source_wiktionary">Wiktionary (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">Daneshjoo Dictionary</string>
<string name="source_wiktionary">Wiktionary — Persian to English (CC BY-SA 3.0)</string>
<string name="source_daneshjoo">Daneshjoo — Persian to English</string>
<string name="dictionary">Dictionary</string>
<string name="source_wiktionary_ur">Wiktionary — Urdu to English (CC BY-SA 3.0)</string>
<string name="source_urwiktionary">Urdu Wiktionary — Urdu to Urdu (CC BY-SA 3.0)</string>
<string name="did_you_mean">Similar words</string>
<string name="source_wiktionary_ar">Wiktionary — Arabic to English (CC BY-SA 3.0)</string>
<string name="oled">Black backgrounds (OLED)</string>
<string name="oled_note">Applies to the dark themes; saves power on OLED screens</string>
<string name="play_recitation">Play recitation</string>
<string name="pause">Pause</string>
</resources>

View File

@ -1,6 +1,10 @@
package com.ganjoor.android
import com.ganjoor.android.data.CatEntry
import com.ganjoor.android.data.Category
import com.ganjoor.android.data.Crumb
import com.ganjoor.android.data.PoemRef
import com.ganjoor.android.data.orderedEntries
import com.ganjoor.android.data.Verse
import com.ganjoor.android.data.breadcrumbs
import com.ganjoor.android.data.parentUrl
@ -125,3 +129,54 @@ class ParentUrlTest {
assertEquals("/hafez", parentUrl("/hafez/ghazal/"))
}
}
class CategoryOrderTest {
private fun cat(chapters: List<String>, poems: List<String>) = Category(
id = 1,
title = "book",
childCats = chapters.mapIndexed { i, t -> Category(id = 100 + i, title = t) },
poems = poems.mapIndexed { i, t -> PoemRef(id = 200 + i, title = t) },
)
private fun titles(category: Category) = orderedEntries(category).map {
when (it) {
is CatEntry.Chapter -> it.category.title
is CatEntry.Poem -> it.poem.title
}
}
@Test
fun `a preface comes before the chapters, as on ganjoor net`() {
// Golestan: دیباچه then the eight باب
val golestan = cat(listOf("باب اول", "باب دوم"), listOf("دیباچه"))
assertEquals(listOf("دیباچه", "باب اول", "باب دوم"), titles(golestan))
}
@Test
fun `other poems stay after the chapters`() {
// Hafez: مقدّمه, then the collections, then مثنوی and ساقی‌نامه
val hafez = cat(
chapters = listOf("غزلیات", "قطعات"),
poems = listOf("مثنوی (الا ای آهوی وحشی)", "ساقی‌نامه", "مقدّمهٔ جمع‌آورندهٔ دیوان حافظ"),
)
assertEquals(
listOf("مقدّمهٔ جمع‌آورندهٔ دیوان حافظ", "غزلیات", "قطعات", "مثنوی (الا ای آهوی وحشی)", "ساقی‌نامه"),
titles(hafez),
)
}
@Test
fun `diacritics in a preface title don't hide it`() {
// مقدّمه carries a shadda the plain spelling doesn't
assertEquals(listOf("مقدّمه", "باب اول"), titles(cat(listOf("باب اول"), listOf("مقدّمه"))))
}
@Test
fun `a category with no poems is left exactly as it is`() {
val masnavi = cat(listOf("دفتر اول", "دفتر دوم", "دفتر سوم"), emptyList())
assertEquals(listOf("دفتر اول", "دفتر دوم", "دفتر سوم"), titles(masnavi))
}
}

View File

@ -1,6 +1,7 @@
package com.ganjoor.android
import com.ganjoor.android.data.affixes
import com.ganjoor.android.data.letterOverlap
import com.ganjoor.android.data.normalise
import com.ganjoor.android.data.wordAt
import org.junit.Assert.assertEquals
@ -79,3 +80,58 @@ class ArabicArticleTest {
assertTrue(affixes("الناس").contains("ناس"))
}
}
class PersianMorphologyTest {
@Test
fun `enclitic pronouns glued onto a verb are stripped`() {
assertTrue(affixes("آیدت").contains("آید"))
assertTrue(affixes("باشدش").contains("باشد"))
assertTrue(affixes("تربتش").contains("تربت"))
}
@Test
fun `a prefix and a negation together still reach the verb`() {
// برنیاید = بر + ن + یاید; one pass would stop at نیاید
assertTrue(affixes("برنیاید").contains("یاید"))
}
@Test
fun `plural and object markers still work`() {
assertTrue(affixes("دلها").contains("دل"))
assertTrue(affixes("مارا").contains("ما"))
}
@Test
fun `stripping never produces a single letter`() {
assertTrue(affixes("شان").none { it.length < 2 })
assertTrue(affixes("بها").none { it.length < 2 })
}
}
class LetterOverlapTest {
@Test
fun `an identical word overlaps completely`() {
assertEquals(1f, letterOverlap("عشق", "عشق"), 0.001f)
}
@Test
fun `a suffixed form still scores high against its stem`() {
assertTrue(letterOverlap("مشکل", "مشکلها") > 0.6f)
}
@Test
fun `sharing only a first letter scores low`() {
assertTrue(letterOverlap("عشق", "عبادتگاه") < 0.4f)
}
@Test
fun `letters are counted once each, not by presence alone`() {
// ااا against ا shares one letter, not three
assertEquals(1f / 3f, letterOverlap("ااا", "ا"), 0.001f)
}
@Test
fun `an empty word never matches`() {
assertEquals(0f, letterOverlap("", "عشق"), 0.001f)
}
}

View File

@ -19,6 +19,29 @@ bundled rather than merely linked.
Nothing in Ganjoor's repositories restricts AI-assisted use.
## Dictionary
Five sources, each row in the database tagged with the one it came from so the app can name it
and so any of them can be dropped without rebuilding the others.
| Source | Direction | Licence |
|---|---|---|
| [Wiktionary](https://en.wiktionary.org) | Persian → English | CC BY-SA 3.0 |
| [Wiktionary](https://en.wiktionary.org) | Urdu → English | CC BY-SA 3.0 |
| [Urdu Wiktionary](https://ur.wiktionary.org) | Urdu → Urdu | CC BY-SA 3.0 |
| [Wiktionary](https://en.wiktionary.org) | Arabic → English | CC BY-SA 3.0 |
| [Daneshjoo](https://github.com/0xdolan/Daneshjoo) | Persian → English | Repository states MIT |
Pronunciation — IPA with the variety it belongs to, and Urdu Wiktionary's vowelled spelling and
syllable split — comes from the same Wiktionary exports and carries the same licence.
Because four of the five are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**.
Attribution is shown in the app beside every definition. See [`../tools/README.md`](../tools/README.md)
to rebuild it.
The Daneshjoo entry is worth a caveat: the repository states MIT, but the underlying lexicon is
a published Iranian dictionary, so that relicensing is worth verifying before relying on it.
## Fonts
| Font | Copyright | Licence | File |

View File

@ -30,6 +30,39 @@ tables, etymology templates, IPA and descendants. The build keeps the definition
form→lemma index (149,589 pairs) and drops the rest. That index is what resolves conjugated
verbs: `افتاد → افتادن`, `بگشاید → گشودن`, `دانند → دانستن`.
## Urdu sources
```sh
curl -L -o ur.jsonl https://kaikki.org/dictionary/Urdu/kaikki.org-dictionary-Urdu.jsonl
curl -L -o urwikt.xml.bz2 \
https://dumps.wikimedia.org/urwiktionary/latest/urwiktionary-latest-pages-articles.xml.bz2
bunzip2 -k urwikt.xml.bz2
```
The first gives Urdu headwords glossed in English. The second is Urdu Wiktionary itself, the only
source here whose definitions are written **in Urdu** — thin (around 3,100 usable entries out of
31,000 pages, many being stubs), but for a word it does carry an Urdu reader is better served by
it than by a translation into English.
## Arabic
```sh
curl -L -o ar.jsonl https://kaikki.org/dictionary/Arabic/kaikki.org-dictionary-Arabic.jsonl
```
521 MB, almost all of it the inflection index, and that index is the point: the Arabic quoted
inside Persian verse is conjugated, so السّاقی, الناس, تَلْقَ and تَهْوی only reach a definition
through it. 36,627 entries and 819,608 new form pairs for about 59 MB of database.
## Why there is no Persian-to-Urdu
Wiktionary's Persian entries carry no translations at all — the translation tables live only on
English pages, in a 3.3 GB export. Going Persian to Urdu would mean pivoting through an English
sense, and a sample of that file projects only about 6,900 Persian words with any Urdu
equivalent, most of them modern dictionary vocabulary rather than the language of the poems. The
definitions written in Urdu therefore come from Urdu Wiktionary directly, and are preferred over
the English ones wherever they exist.
## Licences
Each row carries its `source`, so attribution stays accurate and either source can be dropped

View File

@ -1,5 +1,11 @@
"""Builds the app's dictionary from Wiktionary (CC BY-SA 3.0) and Daneshjoo (MIT).
Three sources, each row tagged so the app can say which one answered and in which language:
wiktionary-fa Persian headwords, English definitions
daneshjoo Persian headwords, English definitions
wiktionary-ur Urdu headwords, English definitions
Wiktionary's export is 93 MB of linguistic metadata around 0.94 MB of definitions, so this keeps
the definitions and the form->lemma index and discards the rest. The `source` column is what
keeps the attribution honest and lets either source be dropped later.
@ -23,9 +29,20 @@ c = sqlite3.connect(db)
c.executescript("""
CREATE TABLE entry (word TEXT NOT NULL, display TEXT NOT NULL, gloss TEXT NOT NULL, source TEXT NOT NULL);
CREATE TABLE form (form TEXT NOT NULL, lemma TEXT NOT NULL);
CREATE TABLE pron (word TEXT NOT NULL, text TEXT NOT NULL, label TEXT NOT NULL, source TEXT NOT NULL);
""")
entries, forms = [], set()
entries, forms, prons = [], set(), set()
def collect_sounds(word, entry, source):
"""IPA with whatever dialect it belongs to. Classical Persian matters most here: this is an
app for poetry written a long time before modern Tehrani vowels."""
for sound in entry.get('sounds') or []:
ipa = (sound.get('ipa') or '').strip()
if not ipa:
continue
tags = [t for t in (sound.get('tags') or []) if t not in ('formal',)]
prons.add((normalise(word), ipa, ' '.join(tags), source))
for line in open('fa.jsonl', encoding='utf-8'):
try: e = json.loads(line)
@ -36,13 +53,54 @@ for line in open('fa.jsonl', encoding='utf-8'):
if gs:
pos = e.get('pos') or ''
gloss = '; '.join(dict.fromkeys(gs))[:600]
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary'))
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa'))
collect_sounds(word, e, 'wiktionary-fa')
for f in e.get('forms', []):
t = f.get('form')
if t and t != word and not t.startswith('-') and len(t) > 1:
forms.add((normalise(t), normalise(word)))
print(f"wiktionary: {len(entries)} entries, {len(forms)} forms")
print(f"wiktionary-fa: {len(entries)} entries, {len(forms)} forms")
# Urdu shares a great deal of vocabulary with Persian, so these entries answer words the
# Persian sources miss. Headwords are Urdu; the definitions are still English.
n_fa = len(entries)
for line in open('ur.jsonl', encoding='utf-8'):
try: e = json.loads(line)
except Exception: continue
word = e.get('word')
if not word: continue
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
if gs:
pos = e.get('pos') or ''
gloss = '; '.join(dict.fromkeys(gs))[:600]
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur'))
collect_sounds(word, e, 'wiktionary-ur')
for f in e.get('forms', []):
t = f.get('form')
if t and t != word and not t.startswith('-') and len(t) > 1:
forms.add((normalise(t), normalise(word)))
print(f"wiktionary-ur: {len(entries) - n_fa} entries, {len(forms)} forms total")
# Arabic, for the lines classical Persian quotes outright — Hafez opens with one.
if os.path.exists('ar.jsonl'):
n_ar, f_ar = len(entries), len(forms)
for line in open('ar.jsonl', encoding='utf-8'):
try: e = json.loads(line)
except Exception: continue
word = e.get('word')
if not word: continue
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
if gs:
pos = e.get('pos') or ''
gloss = '; '.join(dict.fromkeys(gs))[:600]
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar'))
collect_sounds(word, e, 'wiktionary-ar')
for f in e.get('forms', []):
t = f.get('form')
if t and t != word and not t.startswith('-') and len(t) > 1:
forms.add((normalise(t), normalise(word)))
print(f"wiktionary-ar: {len(entries) - n_ar} entries, {len(forms) - f_ar} new forms")
tag = re.compile(r'<[^>]+>')
n0 = len(entries)
@ -58,11 +116,57 @@ for k, v in MDX('daneshjoo.mdx').items():
entries.append((normalise(word), word, txt[:600], 'daneshjoo'))
print(f"daneshjoo : {len(entries) - n0} entries")
# Urdu Wiktionary, the only source here whose definitions are written in Urdu rather than
# English. Thin — a few thousand usable entries, many pages being stubs — but for a word it
# does carry, an Urdu reader is better served by it than by a translation into English.
if os.path.exists('urwikt.xml'):
import html as _html
raw = open('urwikt.xml', encoding='utf-8', errors='ignore').read()
n_ur = len(entries)
for title, ns, body in re.findall(
r'<title>(.*?)</title>.*?<ns>(\d+)</ns>.*?<text[^>]*>(.*?)</text>', raw, re.S
):
if ns != '0' or not re.match(r'^[\u0600-\u06FF]', title):
continue
t = re.sub(r'\{\{[^}]*\}\}', ' ', body)
t = re.sub(r'\[\[([^\]|]*\|)?([^\]]*)\]\]', r'\2', t)
t = re.sub(r"'{2,}|<[^>]+>", '', _html.unescape(t))
section = re.search(r'==\s*معانی\s*==(.*?)(?:\n==|\Z)', t, re.S)
lines = (section.group(1) if section else
'\n'.join(l.strip(' #') for l in t.split('\n') if l.strip().startswith('#')))
kept = []
for line in lines.split('\n'):
line = re.sub(r'^\d+\.\s*', '', line.strip())
# ؎ introduces a verse citation, and "ref"/a year starts the source note; the
# definition itself is what comes before either.
if line.startswith('؎') or re.match(r'^\(?\s*\d{3,4}ء', line):
break
line = re.split(r'\bref\b|؎', line)[0].strip()
if not line or not re.search(r'[\u0600-\u06FF]', line):
continue
kept.append(line)
if len(' '.join(kept)) > 220:
break
gloss = ' '.join(kept).strip()[:300]
if len(gloss) > 3:
entries.append((normalise(title), title, gloss, 'urwiktionary'))
# {عِشْق} is the fully vowelled spelling; "عِش + قوں" is the syllable split
vowelled = re.search(r'\{([\u0600-\u06FF\u064B-\u0652 ]{2,40})\}', t)
if vowelled:
prons.add((normalise(title), vowelled.group(1).strip(), 'اردو', 'urwiktionary'))
syllables = re.search(r'\{([\u0600-\u06FF\u064B-\u0652]+(?: \+ [\u0600-\u06FF\u064B-\u0652]+)+[^}]*)\}', t)
if syllables:
prons.add((normalise(title), syllables.group(1).strip(), 'ہجے', 'urwiktionary'))
print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)")
print(f"pronunciations: {len(prons)}")
c.executemany("INSERT INTO pron VALUES (?,?,?,?)", sorted(prons))
c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries)
c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms))
c.executescript("""
CREATE INDEX idx_entry_word ON entry(word);
CREATE INDEX idx_form_form ON form(form);
CREATE INDEX idx_pron_word ON pron(word);
""")
c.commit()
c.execute("VACUUM")