diff --git a/README.md b/README.md index 7c01ebd..ed518a6 100644 --- a/README.md +++ b/README.md @@ -77,6 +77,15 @@ then with an affix stripped, then the parts of a ZWNJ compound. The lemma step i classical verse readable — `افتاد` is only findable as `افتادن`, and Wiktionary ships 149,589 form→lemma pairs that make that possible. +Each entry carries its pronunciation where Wiktionary has one — 101,306 of them, tagged with the +variety, Classical Persian first, because a word in a 14th-century ghazal was not said the way +Tehran says it now. Urdu Wiktionary adds the vowelled spelling and the syllable split in Urdu +script. When nothing matches at all, the sheet offers near words ranked by shared letters. + +Poems that Ganjoor has recordings for show a play button, streamed rather than stored: a famous +ghazal often has a dozen readings, and downloading them would dwarf the poems. It is the one +part of the app that needs a connection, and it simply doesn't appear without one. + See [`tools/README.md`](tools/README.md) to rebuild it, and for the licensing of each source. ## Where the poems come from diff --git a/app/src/main/assets/dictionary.db b/app/src/main/assets/dictionary.db index c6ddab9..92e8958 100644 Binary files a/app/src/main/assets/dictionary.db and b/app/src/main/assets/dictionary.db differ diff --git a/app/src/main/java/com/ganjoor/android/MainActivity.kt b/app/src/main/java/com/ganjoor/android/MainActivity.kt index 6db461a..ea205dc 100644 --- a/app/src/main/java/com/ganjoor/android/MainActivity.kt +++ b/app/src/main/java/com/ganjoor/android/MainActivity.kt @@ -46,7 +46,7 @@ class MainActivity : ComponentActivity() { val systemInDark = resources.configuration.uiMode and Configuration.UI_MODE_NIGHT_MASK == Configuration.UI_MODE_NIGHT_YES window.setBackgroundDrawable( - windowBackground(settings.value.theme, systemInDark).toDrawable() + windowBackground(settings.value.theme, systemInDark, settings.value.oled).toDrawable() ) enableEdgeToEdge() @@ -62,7 +62,7 @@ class MainActivity : ComponentActivity() { // navigates right-to-left whichever UI language is selected. LocalLayoutDirection provides LayoutDirection.Rtl, ) { - GanjoorTheme(settings.value.theme, settings.value.language) { + GanjoorTheme(settings.value.theme, settings.value.language, settings.value.oled) { GanjoorApp() } } diff --git a/app/src/main/java/com/ganjoor/android/data/Dictionary.kt b/app/src/main/java/com/ganjoor/android/data/Dictionary.kt index c58f913..8ce40c9 100644 --- a/app/src/main/java/com/ganjoor/android/data/Dictionary.kt +++ b/app/src/main/java/com/ganjoor/android/data/Dictionary.kt @@ -11,13 +11,19 @@ import java.text.Normalizer /** One definition, and where it came from, so the credit stays attached to the text. */ data class Definition(val word: String, val gloss: String, val source: String) +/** + * How a word sounds. [label] is the variety it belongs to — Classical Persian, Dari, Standard + * Urdu — because a word in a 14th-century ghazal was not said the way Tehran says it now. + */ +data class Pronunciation(val text: String, val label: String) + /** * Word lookup over a bundled SQLite built from Wiktionary (CC BY-SA 3.0) and Daneshjoo. * See `tools/build_dictionary.py`; the asset ships gzipped and is unpacked once on first use. */ object Dictionary { /** Guards against a copy interrupted half-way leaving an unopenable file behind. */ - private const val ASSET_BYTES = 22_298_624L + private const val ASSET_BYTES = 94384128L // Not a .gz: the build packager silently gunzips those and drops the extension, which left // the asset under a different name than the code was opening. @@ -46,6 +52,37 @@ object Dictionary { } } + /** + * How [raw] is pronounced, Classical Persian first: this is an app for poetry written long + * before modern Tehrani vowels, and the IPA's dots and stress marks are the syllable + * breakdown that goes with it. + */ + suspend fun pronunciations(raw: String, limit: Int = 5): List = + withContext(Dispatchers.IO) { + val database = open() ?: return@withContext emptyList() + val word = normalise(raw).takeIf { it.isNotEmpty() } ?: return@withContext emptyList() + database.rawQuery( + """ + SELECT text, label FROM pron WHERE word = ? + ORDER BY CASE + WHEN label LIKE 'Classical%' THEN 0 + WHEN source = 'urwiktionary' THEN 1 + WHEN source = 'wiktionary-fa' THEN 2 + WHEN source = 'wiktionary-ur' THEN 3 + ELSE 4 + END + LIMIT ? + """, + arrayOf(word, limit.toString()), + ).use { cursor -> + buildList { + while (cursor.moveToNext()) { + add(Pronunciation(cursor.getString(0), cursor.getString(1))) + } + } + } + } + /** * Looks a word up, widening the search until something matches: * @@ -69,11 +106,42 @@ object Dictionary { .firstNotNullOfOrNull { direct(database, normalise(it)).ifEmpty { null } } .orEmpty() } + // A selection is usually a phrase rather than a word; fall back to its words. + .ifEmpty { + raw.split(' ', '\n', '\r') + .map { normalise(it) } + .filter { it.length > 1 && it != word } + .firstNotNullOfOrNull { part -> + direct(database, part).ifEmpty { + lemmas(database, part).flatMap { direct(database, it) }.ifEmpty { null } + } + } + .orEmpty() + } } + /** + * Ordered the way a reader of this app wants to be answered: a definition written in Urdu + * first, because it needs no translating at all, then the Persian sources, then the ones + * keyed on another language. English is what the rest fall back to, so it comes last by + * coming from the sources that sit last. + * + * Arabic is last outright: its forms index is larger than every other source combined, + * which makes it the likeliest to match something by coincidence. + */ private fun direct(database: SQLiteDatabase, word: String): List = database.rawQuery( - "SELECT display, gloss, source FROM entry WHERE word = ? LIMIT 12", + """ + SELECT display, gloss, source FROM entry WHERE word = ? + ORDER BY CASE source + WHEN 'urwiktionary' THEN 0 + WHEN 'wiktionary-fa' THEN 1 + WHEN 'daneshjoo' THEN 2 + WHEN 'wiktionary-ur' THEN 3 + ELSE 4 + END + LIMIT 12 + """, arrayOf(word), ).use { cursor -> buildList { @@ -83,6 +151,47 @@ object Dictionary { } } + /** + * Headwords that look like [raw], for when nothing matched exactly — a misread letter, an + * unusual spelling, or a word the dictionary simply spells differently. + * + * Candidates come from an index range scan on the first letters, then are ranked by how + * many letters they share with the query. ponytail: shared letters rather than an edit + * distance, which would need the whole table scanned to be worth the extra precision. + */ + suspend fun suggest(raw: String, limit: Int = 6): List = withContext(Dispatchers.IO) { + val database = open() ?: return@withContext emptyList() + val word = normalise(raw).takeIf { it.length > 1 } ?: return@withContext emptyList() + + // Widen the prefix until there is something to rank, but never scan the whole table. + val candidates = generateSequence(minOf(3, word.length - 1)) { (it - 1).takeIf { n -> n >= 1 } } + .map { prefixLength -> byPrefix(database, word.take(prefixLength)) } + .firstOrNull { it.size >= 3 } + ?: return@withContext emptyList() + + candidates + .asSequence() + .filter { it.first != word } + .map { (normalised, display) -> display to letterOverlap(word, normalised) } + .filter { it.second > 0.45f } + .sortedByDescending { it.second } + .map { it.first } + .distinct() + .take(limit) + .toList() + } + + /** Index range scan: everything whose normalised form starts with [prefix]. */ + private fun byPrefix(database: SQLiteDatabase, prefix: String): List> = + database.rawQuery( + "SELECT DISTINCT word, display FROM entry WHERE word >= ? AND word < ? LIMIT 400", + arrayOf(prefix, prefix + '\uFFFF'), + ).use { cursor -> + buildList { + while (cursor.moveToNext()) add(cursor.getString(0) to cursor.getString(1)) + } + } + private fun lemmas(database: SQLiteDatabase, form: String): List = database.rawQuery( "SELECT lemma FROM form WHERE form = ? LIMIT 6", @@ -94,14 +203,29 @@ object Dictionary { private const val ZWNJ = '‌' -private val SUFFIXES = listOf("ها", "اش", "ش", "م", "ت", "را", "ی", "ان") -// "ال" is the Arabic definite article: poems quote Arabic, so السّاقی has to reach ساقی. -private val PREFIXES = listOf("ال", "می", "بر", "ب") +/** + * Longest first, so تربتش strips شـ rather than matching nothing. These are the endings that + * actually turn up in classical verse: plurals, the object marker, and the enclitic pronouns + * that Persian glues onto a verb or noun — آیدت is آید + ت, باشدش is باشد + ش. + */ +private val SUFFIXES = listOf( + "شان", "تان", "مان", "ها", "اش", "ست", "یم", "ید", "ند", "را", "ش", "م", "ت", "ی", "ان", "ه", +) -/** Candidate stems after stripping one common affix. Order matters: longest affix first. */ -internal fun affixes(word: String): List = buildList { - SUFFIXES.forEach { if (word.endsWith(it) && word.length > it.length + 1) add(word.dropLast(it.length)) } - PREFIXES.forEach { if (word.startsWith(it) && word.length > it.length + 1) add(word.drop(it.length)) } +/** "ال" is the Arabic definite article, "ن"/"نمی" negation, "بی" privative. */ +private val PREFIXES = listOf("نمی", "ال", "می", "بی", "بر", "ن", "ب") + +/** + * Candidate stems, one affix deep and then two — برنیاید is بر + ن + یاید, and a single pass + * would never reach the verb. Ordered so the least mangled candidate is tried first. + */ +internal fun affixes(word: String): List { + fun oneStep(w: String) = buildList { + SUFFIXES.forEach { if (w.endsWith(it) && w.length > it.length + 1) add(w.dropLast(it.length)) } + PREFIXES.forEach { if (w.startsWith(it) && w.length > it.length + 1) add(w.drop(it.length)) } + } + val first = oneStep(word) + return (first + first.flatMap(::oneStep)).distinct() } private val HARAKAT = (0x064B..0x0652) + listOf(0x0670, 0x0640) + (0x0610..0x0615) @@ -125,6 +249,17 @@ internal fun normalise(text: String, keepZwnj: Boolean = false): String { return (if (keepZwnj) folded else folded.replace(ZWNJ.toString(), "")).trim() } +/** + * How much of two words' letters coincide — the overlapping letters counted against the longer + * word, so مشکل and مشکلها score high while a word that merely starts the same does not. + */ +internal fun letterOverlap(a: String, b: String): Float { + if (a.isEmpty() || b.isEmpty()) return 0f + val remaining = b.toMutableList() + val shared = a.count { remaining.remove(it) } + return shared.toFloat() / maxOf(a.length, b.length) +} + /** The whole word surrounding [index], for turning a tap into something to look up. */ internal fun wordAt(text: String, index: Int): String? { if (text.isEmpty()) return null diff --git a/app/src/main/java/com/ganjoor/android/data/Ganjoor.kt b/app/src/main/java/com/ganjoor/android/data/Ganjoor.kt index df0dcd0..6766741 100644 --- a/app/src/main/java/com/ganjoor/android/data/Ganjoor.kt +++ b/app/src/main/java/com/ganjoor/android/data/Ganjoor.kt @@ -145,6 +145,18 @@ private data class LiveCat(val poems: List = emptyList()) @Serializable private data class LivePoem(val id: Int = 0, val excerpt: String? = null) +/** A reading of a whole poem, hosted by Ganjoor. */ +@Serializable +data class Recitation( + val id: Int = 0, + val audioTitle: String = "", + val audioArtist: String = "", + val mp3Url: String = "", +) + +@Serializable +private data class LivePoemRecitations(val recitations: List = emptyList()) + private val liveJson = Json { ignoreUnknownKeys = true } @Serializable @@ -166,6 +178,37 @@ fun catPath(fullUrl: String) = "poets${fullUrl.trimEnd('/')}/_cat.json" fun poemPath(fullUrl: String) = "poets${fullUrl.trimEnd('/')}.json" +/** A row in a category listing: either a chapter to open, or a poem to read. */ +sealed interface CatEntry { + data class Chapter(val category: Category) : CatEntry + data class Poem(val poem: PoemRef) : CatEntry +} + +private val PREFACE_TITLES = listOf("دیباچه", "مقدمه", "سرآغاز", "پیشگفتار", "آغاز") + +/** + * Orders a category the way ganjoor.net does: a book's own preface first, then its chapters, + * then whatever other poems sit directly under it. Golestan's دیباچه belongs above the eight + * باب, not below them; Hafez's مقدّمه above his five collections, with مثنوی and ساقی‌نامه after. + * + * Ganjoor decides this with each poem's MixedModeOrder — 1 sorts a poem above the chapters, 0 + * below — but that field isn't in the exported `_cat.json`, only on the live API's per-poem + * record, which would be one request per poem. + * + * ponytail: so prefaces are recognised by title instead. Adding MixedModeOrder to the Poems + * entries in ganjoor-data would make this exact; until then a book whose preface is named + * something unusual still lands after its chapters. + */ +fun orderedEntries(category: Category): List { + val (prefaces, rest) = category.poems.partition { poem -> + val title = normalise(poem.title).trimStart() + PREFACE_TITLES.any { title.startsWith(it) } + } + return prefaces.map(CatEntry::Poem) + + category.childCats.map(CatEntry::Chapter) + + rest.map(CatEntry::Poem) +} + /** One step of a poem's path. [url] is null for the poem itself, which is already open. */ data class Crumb(val label: String, val url: String?) @@ -324,6 +367,27 @@ object Ganjoor { } } + /** + * Readings of a poem, by the people who recorded them for ganjoor.net. + * + * Streamed, never stored: the files are a few hundred kilobytes each and there are often a + * dozen readings of a famous ghazal, so downloading them all would dwarf the poems. Offline + * mode therefore has none of this, which is honest — a recording is the one thing here that + * genuinely needs the network. + */ + suspend fun recitations(poemId: Int): List = withContext(Dispatchers.IO) { + if (offline || poemId == 0) return@withContext emptyList() + val url = liveBase.newBuilder() + .addPathSegments("api/ganjoor/poem/$poemId") + .addQueryParameter("recitations", "true") + .addQueryParameter("verseDetails", "false") + .build() + runCatching { + liveJson.decodeFromString(fetch(url)).recitations + .filter { it.mp3Url.isNotBlank() } + }.getOrDefault(emptyList()) + } + /** * Saves a poet's whole tree — biography, every category and every poem — for offline reading. * Re-running it is cheap: anything already on disk is skipped, which is also how a download diff --git a/app/src/main/java/com/ganjoor/android/ui/AboutScreen.kt b/app/src/main/java/com/ganjoor/android/ui/AboutScreen.kt index c96923c..ca12c06 100644 --- a/app/src/main/java/com/ganjoor/android/ui/AboutScreen.kt +++ b/app/src/main/java/com/ganjoor/android/ui/AboutScreen.kt @@ -64,11 +64,17 @@ private val CREDITS = listOf( ), Credit( "Wiktionary", - "Wiktionary contributors — the word definitions, and the inflected-form index that " + - "finds a conjugated verb's dictionary entry", + "Wiktionary contributors — Persian and Urdu definitions, and the inflected-form " + + "index that finds a conjugated verb's dictionary entry", "CC BY-SA 3.0. The bundled dictionary is therefore also CC BY-SA 3.0.", url = "https://en.wiktionary.org", ), + Credit( + "Urdu Wiktionary", + "Urdu Wiktionary contributors — the definitions written in Urdu rather than English", + "CC BY-SA 3.0. The bundled dictionary is therefore also CC BY-SA 3.0.", + url = "https://ur.wiktionary.org", + ), Credit( "Daneshjoo Dictionary", "Layered under Wiktionary for the words it doesn't carry", diff --git a/app/src/main/java/com/ganjoor/android/ui/CategoryScreen.kt b/app/src/main/java/com/ganjoor/android/ui/CategoryScreen.kt index fb61485..3f408b6 100644 --- a/app/src/main/java/com/ganjoor/android/ui/CategoryScreen.kt +++ b/app/src/main/java/com/ganjoor/android/ui/CategoryScreen.kt @@ -38,9 +38,11 @@ import androidx.compose.ui.res.stringResource import androidx.compose.ui.text.style.TextOverflow import androidx.compose.ui.unit.dp import com.ganjoor.android.R +import com.ganjoor.android.data.CatEntry import com.ganjoor.android.data.Downloads import com.ganjoor.android.data.Ganjoor import com.ganjoor.android.data.Offline +import com.ganjoor.android.data.orderedEntries @OptIn(ExperimentalMaterial3Api::class) @Composable @@ -63,6 +65,8 @@ fun CategoryScreen( } } + val entries = remember(cat) { orderedEntries(cat) } + Scaffold( topBar = { TopAppBar( @@ -94,12 +98,25 @@ fun CategoryScreen( cat.description?.takeIf { it.isNotBlank() }?.let { description -> item { Description(description) } } - items(cat.childCats, key = { "c${it.id}" }) { child -> - NavRow(child.title, isCategory = true) { onCategory(child.fullUrl) } - } - items(cat.poems, key = { "p${it.id}" }) { poem -> - NavRow(poem.title, isCategory = false, excerpt = excerpts[poem.id]) { - onPoem(poem.fullUrl) + items( + items = entries, + key = { entry -> + when (entry) { + is CatEntry.Chapter -> "c${entry.category.id}" + is CatEntry.Poem -> "p${entry.poem.id}" + } + }, + ) { entry -> + when (entry) { + is CatEntry.Chapter -> NavRow(entry.category.title, isCategory = true) { + onCategory(entry.category.fullUrl) + } + + is CatEntry.Poem -> NavRow( + title = entry.poem.title, + isCategory = false, + excerpt = excerpts[entry.poem.id], + ) { onPoem(entry.poem.fullUrl) } } } } diff --git a/app/src/main/java/com/ganjoor/android/ui/PoemScreen.kt b/app/src/main/java/com/ganjoor/android/ui/PoemScreen.kt index 633ef1a..66415dc 100644 --- a/app/src/main/java/com/ganjoor/android/ui/PoemScreen.kt +++ b/app/src/main/java/com/ganjoor/android/ui/PoemScreen.kt @@ -105,8 +105,14 @@ fun PoemScreen( ) } ) { insets -> - // Free-form selection for copying any span; the per-couplet actions below are for - // saving a passage with the reference attached, which a raw copy would lose. + // Free-form selection for copying any span; tapping a word looks it up, and the + // per-couplet actions save a passage with the reference attached, which a raw copy + // would lose. + // + // ponytail: no dictionary entry in the selection toolbar. Compose 1.10 stopped + // routing SelectionContainer through LocalTextToolbar — a custom TextToolbar is + // simply never asked to show — and the replacement, foundation's contextmenu + // package, is internal. Revisit when that becomes public API. SelectionContainer { LazyColumn( modifier = Modifier.fillMaxSize(), @@ -120,6 +126,7 @@ fun PoemScreen( item { Column(Modifier.padding(bottom = 12.dp)) { Breadcrumbs(poem.fullTitle, poem.fullUrl.ifBlank { fullUrl }, onCategory) + RecitationPlayer(poem.id) poem.metre?.rhythm?.let { rhythm -> Text( text = rhythm, diff --git a/app/src/main/java/com/ganjoor/android/ui/ReadingSettings.kt b/app/src/main/java/com/ganjoor/android/ui/ReadingSettings.kt index 1a5bd69..c5c99a4 100644 --- a/app/src/main/java/com/ganjoor/android/ui/ReadingSettings.kt +++ b/app/src/main/java/com/ganjoor/android/ui/ReadingSettings.kt @@ -78,6 +78,15 @@ fun ReadingSettingsSheet(onDismiss: () -> Unit) { settings.update { it.copy(theme = mode) } } + // Only means anything on a dark theme, so it sits with them and says so. + Toggle( + title = R.string.oled, + note = R.string.oled_note, + checked = prefs.oled, + ) { on -> + settings.update { it.copy(oled = on) } + } + Label(R.string.font) Chips(ReadingFont.entries, prefs.font, { stringResource(it.label) }) { font -> settings.update { it.copy(font = font) } diff --git a/app/src/main/java/com/ganjoor/android/ui/RecitationPlayer.kt b/app/src/main/java/com/ganjoor/android/ui/RecitationPlayer.kt new file mode 100644 index 0000000..6513ab5 --- /dev/null +++ b/app/src/main/java/com/ganjoor/android/ui/RecitationPlayer.kt @@ -0,0 +1,141 @@ +package com.ganjoor.android.ui + +import android.media.AudioAttributes +import android.media.MediaPlayer +import androidx.compose.foundation.layout.Arrangement +import androidx.compose.foundation.layout.Row +import androidx.compose.foundation.layout.fillMaxWidth +import androidx.compose.foundation.layout.padding +import androidx.compose.foundation.layout.size +import androidx.compose.material.icons.Icons +import androidx.compose.material.icons.filled.PlayArrow +import androidx.compose.material3.CircularProgressIndicator +import androidx.compose.material3.DropdownMenu +import androidx.compose.material3.DropdownMenuItem +import androidx.compose.material3.Icon +import androidx.compose.material3.IconButton +import androidx.compose.material3.MaterialTheme +import androidx.compose.material3.Text +import androidx.compose.material3.TextButton +import androidx.compose.runtime.Composable +import androidx.compose.runtime.DisposableEffect +import androidx.compose.runtime.getValue +import androidx.compose.runtime.mutableIntStateOf +import androidx.compose.runtime.mutableStateOf +import androidx.compose.runtime.produceState +import androidx.compose.runtime.remember +import androidx.compose.runtime.setValue +import androidx.compose.ui.Alignment +import androidx.compose.ui.Modifier +import androidx.compose.ui.res.painterResource +import androidx.compose.ui.res.stringResource +import androidx.compose.ui.text.style.TextOverflow +import androidx.compose.ui.unit.dp +import com.ganjoor.android.R +import com.ganjoor.android.data.Ganjoor +import com.ganjoor.android.data.Recitation + +/** + * Plays a reading of the poem, streamed from Ganjoor. + * + * ponytail: the platform's MediaPlayer rather than ExoPlayer — one URL, play and pause, no + * playlist or seeking to justify a media library. Nothing is cached, so this is the one part of + * the app that needs a connection; it simply doesn't appear when there is no reading or no + * network. + */ +@Composable +fun RecitationPlayer(poemId: Int) { + val recitations by produceState(emptyList(), poemId) { + value = Ganjoor.recitations(poemId) + } + if (recitations.isEmpty()) return + + var chosen by remember(poemId) { mutableIntStateOf(0) } + var playing by remember(poemId) { mutableStateOf(false) } + var loading by remember(poemId) { mutableStateOf(false) } + var picking by remember { mutableStateOf(false) } + + val player = remember { + MediaPlayer().apply { + setAudioAttributes( + AudioAttributes.Builder() + .setUsage(AudioAttributes.USAGE_MEDIA) + .setContentType(AudioAttributes.CONTENT_TYPE_MUSIC) + .build() + ) + } + } + // A reading left playing when the screen goes would keep the whole poem in memory. + DisposableEffect(player) { onDispose { runCatching { player.release() } } } + + val recitation = recitations.getOrNull(chosen) ?: return + // Changing reader stops whatever was playing, so the two never overlap. + DisposableEffect(recitation.mp3Url) { + runCatching { player.reset() } + playing = false + loading = false + onDispose { } + } + + fun toggle() { + if (playing) { + runCatching { player.pause() } + playing = false + return + } + if (player.currentPosition > 0) { + runCatching { player.start() }.onSuccess { playing = true } + return + } + loading = true + runCatching { + player.reset() + player.setDataSource(recitation.mp3Url) + player.setOnPreparedListener { it.start(); playing = true; loading = false } + player.setOnCompletionListener { playing = false } + player.setOnErrorListener { _, _, _ -> playing = false; loading = false; true } + player.prepareAsync() + }.onFailure { loading = false } + } + + Row( + modifier = Modifier.fillMaxWidth().padding(bottom = 8.dp), + verticalAlignment = Alignment.CenterVertically, + horizontalArrangement = Arrangement.spacedBy(4.dp), + ) { + IconButton(onClick = ::toggle) { + when { + loading -> CircularProgressIndicator(Modifier.size(20.dp), strokeWidth = 2.dp) + // Core Material icons ship no pause glyph, and the extended set is 4 MB for one. + playing -> Icon(painterResource(R.drawable.ic_pause), stringResource(R.string.pause)) + else -> Icon(Icons.Default.PlayArrow, stringResource(R.string.play_recitation)) + } + } + TextButton( + onClick = { if (recitations.size > 1) picking = true }, + modifier = Modifier.weight(1f, fill = false), + ) { + Text( + text = recitation.audioArtist.ifBlank { stringResource(R.string.play_recitation) }, + style = MaterialTheme.typography.labelLarge, + maxLines = 1, + overflow = TextOverflow.Ellipsis, + ) + } + if (recitations.size > 1) { + Text( + text = "${chosen + 1}/${recitations.size}", + style = MaterialTheme.typography.labelSmall, + color = MaterialTheme.colorScheme.onSurfaceVariant, + ) + } + DropdownMenu(expanded = picking, onDismissRequest = { picking = false }) { + recitations.forEachIndexed { index, item -> + DropdownMenuItem( + text = { Text(item.audioArtist.ifBlank { item.audioTitle }) }, + onClick = { chosen = index; picking = false }, + ) + } + } + } +} diff --git a/app/src/main/java/com/ganjoor/android/ui/Settings.kt b/app/src/main/java/com/ganjoor/android/ui/Settings.kt index ccccc2a..33a7466 100644 --- a/app/src/main/java/com/ganjoor/android/ui/Settings.kt +++ b/app/src/main/java/com/ganjoor/android/ui/Settings.kt @@ -18,9 +18,6 @@ enum class ThemeMode(@StringRes val label: Int) { Dark(R.string.theme_dark), Sepia(R.string.theme_sepia), SepiaDark(R.string.theme_sepia_dark), - - /** Pure black, so OLED panels can switch the pixels off entirely. */ - Black(R.string.theme_black), } enum class ReadingFont(@StringRes val label: Int) { @@ -58,6 +55,8 @@ data class Prefs( val showSummaries: Boolean = false, val language: Language = Language.Fa, val offline: Boolean = false, + /** True black backgrounds, applied to whichever dark theme is in use. */ + val oled: Boolean = false, val poetSort: PoetSort = PoetSort.Default, ) @@ -74,6 +73,8 @@ class Settings(context: Context) { showSummaries = prefs.getBoolean("showSummaries", false), language = languageOrDefault(prefs.getString(KEY_LANGUAGE, null)), offline = prefs.getBoolean("offline", false), + // "Black" used to be a sixth theme; it is a flag on the dark ones now. + oled = prefs.getBoolean("oled", prefs.getString("theme", null) == "Black"), poetSort = enumOrDefault(prefs.getString("poetSort", null), PoetSort.Default), ) ) @@ -91,6 +92,7 @@ class Settings(context: Context) { putBoolean("showSummaries", p.showSummaries) putString(KEY_LANGUAGE, p.language.tag) putBoolean("offline", p.offline) + putBoolean("oled", p.oled) putString("poetSort", p.poetSort.name) } } diff --git a/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt b/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt index 49c88d8..cdc3678 100644 --- a/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt +++ b/app/src/main/java/com/ganjoor/android/ui/WordSheet.kt @@ -2,22 +2,28 @@ package com.ganjoor.android.ui import androidx.compose.foundation.layout.Arrangement import androidx.compose.foundation.layout.Column +import androidx.compose.foundation.layout.Row import androidx.compose.foundation.layout.fillMaxWidth import androidx.compose.foundation.layout.navigationBarsPadding import androidx.compose.foundation.layout.padding import androidx.compose.foundation.rememberScrollState import androidx.compose.foundation.verticalScroll +import androidx.compose.foundation.layout.FlowRow import androidx.compose.material3.CircularProgressIndicator import androidx.compose.material3.ExperimentalMaterial3Api import androidx.compose.material3.HorizontalDivider import androidx.compose.material3.MaterialTheme import androidx.compose.material3.ModalBottomSheet +import androidx.compose.material3.SuggestionChip import androidx.compose.material3.Text import androidx.compose.runtime.Composable import androidx.compose.runtime.CompositionLocalProvider import androidx.compose.runtime.getValue import androidx.compose.runtime.mutableStateOf import androidx.compose.runtime.produceState +import androidx.compose.runtime.remember +import androidx.compose.runtime.setValue +import androidx.compose.ui.Alignment import androidx.compose.ui.Modifier import androidx.compose.ui.platform.LocalLayoutDirection import androidx.compose.ui.res.stringResource @@ -26,6 +32,7 @@ import androidx.compose.ui.unit.dp import com.ganjoor.android.R import com.ganjoor.android.data.Definition import com.ganjoor.android.data.Dictionary +import com.ganjoor.android.data.Pronunciation import com.ganjoor.android.ui.theme.readingStyle /** English prose inside an otherwise right-to-left sheet. */ @@ -34,12 +41,43 @@ private fun LeftToRight(content: @Composable () -> Unit) { CompositionLocalProvider(LocalLayoutDirection provides LayoutDirection.Ltr, content = content) } +/** Lays a definition out the way its own script reads. */ +@Composable +private fun InDirectionOf(text: String, content: @Composable () -> Unit) { + val arabicScript = text.count { it in '\u0600'..'\u06FF' } + val latin = text.count { it in 'A'..'Z' || it in 'a'..'z' } + CompositionLocalProvider( + LocalLayoutDirection provides + if (arabicScript > latin) LayoutDirection.Rtl else LayoutDirection.Ltr, + content = content, + ) +} + +/** Which dictionary answered, and in which language pair. */ +private fun sourceLabel(source: String) = when (source) { + "wiktionary-fa" -> R.string.source_wiktionary + "wiktionary-ur" -> R.string.source_wiktionary_ur + "urwiktionary" -> R.string.source_urwiktionary + "wiktionary-ar" -> R.string.source_wiktionary_ar + else -> R.string.source_daneshjoo +} + /** What the dictionary knows about a tapped word. */ @OptIn(ExperimentalMaterial3Api::class) @Composable fun WordSheet(word: String, onDismiss: () -> Unit) { val prefs = LocalSettings.current.value - val definitions by produceState?>(null, word) { value = Dictionary.lookup(word) } + // A suggestion replaces what is being looked up, so the sheet can be followed like a trail. + var current by remember(word) { mutableStateOf(word) } + val definitions by produceState?>(null, current) { + value = Dictionary.lookup(current) + } + val sounds by produceState(emptyList(), current) { + value = Dictionary.pronunciations(current) + } + val suggestions by produceState(emptyList(), current, definitions) { + value = if (definitions?.isEmpty() == true) Dictionary.suggest(current) else emptyList() + } ModalBottomSheet(onDismissRequest = onDismiss) { Column( @@ -52,47 +90,90 @@ fun WordSheet(word: String, onDismiss: () -> Unit) { ) { // The headword in the reading font, at reading size: it is a line of poetry, after all. Text( - text = word, + text = current, style = readingStyle(prefs.font, prefs.fontSize, prefs.fontWeight.weight), modifier = Modifier.fillMaxWidth(), ) + if (sounds.isNotEmpty()) { + // One direction for the whole block, and one pronunciation per line. Flowing + // them side by side put Latin IPA and Urdu spelling in the same right-to-left + // run, which reordered the chips and left each label under someone else's value. + LeftToRight { + Column(verticalArrangement = Arrangement.spacedBy(2.dp)) { + sounds.forEach { sound -> + Row( + modifier = Modifier.fillMaxWidth(), + horizontalArrangement = Arrangement.spacedBy(10.dp), + verticalAlignment = Alignment.CenterVertically, + ) { + Text( + text = sound.text, + style = MaterialTheme.typography.bodyMedium, + modifier = Modifier.weight(1f, fill = false), + ) + if (sound.label.isNotBlank()) { + Text( + text = sound.label, + style = MaterialTheme.typography.labelSmall, + color = MaterialTheme.colorScheme.onSurfaceVariant, + ) + } + } + } + } + } + } HorizontalDivider() when { definitions == null -> CircularProgressIndicator(Modifier.padding(vertical = 16.dp)) - definitions!!.isEmpty() -> LeftToRight { - Text( - text = stringResource(R.string.no_definition), - style = MaterialTheme.typography.bodyMedium, - color = MaterialTheme.colorScheme.onSurfaceVariant, - modifier = Modifier.fillMaxWidth(), - ) + definitions!!.isEmpty() -> Column( + verticalArrangement = Arrangement.spacedBy(8.dp), + ) { + LeftToRight { + Text( + text = stringResource(R.string.no_definition), + style = MaterialTheme.typography.bodyMedium, + color = MaterialTheme.colorScheme.onSurfaceVariant, + modifier = Modifier.fillMaxWidth(), + ) + } + if (suggestions.isNotEmpty()) { + Text( + text = stringResource(R.string.did_you_mean), + style = MaterialTheme.typography.titleSmall, + color = MaterialTheme.colorScheme.primary, + ) + FlowRow(horizontalArrangement = Arrangement.spacedBy(8.dp)) { + suggestions.forEach { suggestion -> + SuggestionChip( + onClick = { current = suggestion }, + label = { Text(suggestion) }, + ) + } + } + } } else -> definitions!!.forEach { definition -> Column(Modifier.fillMaxWidth()) { // The headword actually matched, which may be the lemma rather than the // word as it appears in the line. Persian, so it stays right-to-left. - if (definition.word != word) { + if (definition.word != current) { Text( text = definition.word, style = MaterialTheme.typography.titleSmall, color = MaterialTheme.colorScheme.primary, ) } - // The definitions are English; right-aligning them reads badly. - LeftToRight { + // Most definitions are English and right-aligning them reads badly; + // the Urdu ones are right-to-left like the rest of the app. + InDirectionOf(definition.gloss) { Column(Modifier.fillMaxWidth()) { Text(definition.gloss, style = MaterialTheme.typography.bodyMedium) Text( - text = stringResource( - if (definition.source == "wiktionary") { - R.string.source_wiktionary - } else { - R.string.source_daneshjoo - } - ), + text = stringResource(sourceLabel(definition.source)), style = MaterialTheme.typography.labelSmall, color = MaterialTheme.colorScheme.onSurfaceVariant, ) diff --git a/app/src/main/java/com/ganjoor/android/ui/theme/Theme.kt b/app/src/main/java/com/ganjoor/android/ui/theme/Theme.kt index 49bb34a..3be6baf 100644 --- a/app/src/main/java/com/ganjoor/android/ui/theme/Theme.kt +++ b/app/src/main/java/com/ganjoor/android/ui/theme/Theme.kt @@ -2,12 +2,14 @@ package com.ganjoor.android.ui.theme import android.app.Activity import androidx.compose.foundation.isSystemInDarkTheme +import androidx.compose.material3.ColorScheme import androidx.compose.material3.MaterialTheme import androidx.compose.material3.darkColorScheme import androidx.compose.material3.lightColorScheme import androidx.compose.runtime.Composable import androidx.compose.runtime.SideEffect import androidx.compose.ui.graphics.Color +import androidx.compose.ui.graphics.lerp import androidx.compose.ui.graphics.toArgb import androidx.compose.ui.platform.LocalView import androidx.core.graphics.drawable.toDrawable @@ -66,36 +68,6 @@ private val DarkScheme = darkColorScheme( outline = Color(0xFF899393), ) -/** - * True black, for OLED panels: a black pixel is an unlit pixel, so a night-time reading session - * costs noticeably less battery than the regular dark theme's dark grey. Surfaces step up in - * near-black greys so cards and sheets stay distinguishable without lighting the whole screen. - */ -private val BlackScheme = darkColorScheme( - primary = Color(0xFF80D4DA), - onPrimary = Color(0xFF00363A), - primaryContainer = Color(0xFF004F53), - onPrimaryContainer = Color(0xFF9CF1F6), - secondary = Color(0xFFFFB873), - onSecondary = Color(0xFF4A2800), - secondaryContainer = Color(0xFF693C00), - onSecondaryContainer = Color(0xFFFFDCBE), - background = Color(0xFF000000), - onBackground = Color(0xFFE3E3E3), - surface = Color(0xFF000000), - onSurface = Color(0xFFE3E3E3), - surfaceVariant = Color(0xFF1C1C1C), - onSurfaceVariant = Color(0xFFBDBDBD), - outline = Color(0xFF6E6E6E), - surfaceBright = Color(0xFF262626), - surfaceDim = Color(0xFF000000), - surfaceContainerLowest = Color(0xFF000000), - surfaceContainerLow = Color(0xFF0A0A0A), - surfaceContainer = Color(0xFF101010), - surfaceContainerHigh = Color(0xFF1A1A1A), - surfaceContainerHighest = Color(0xFF242424), -) - // Aged paper, for long reading sessions. private val SepiaScheme = lightColorScheme( primary = Color(0xFF7A4E24), @@ -153,28 +125,61 @@ private val SepiaDarkScheme = darkColorScheme( * so the gap between the window appearing and the first frame matches the theme instead of * flashing the platform's white. A static XML theme can't express sepia or OLED, hence this. */ -fun windowBackground(mode: ThemeMode, systemInDark: Boolean): Int = when (mode) { - ThemeMode.Light -> LightScheme - ThemeMode.Dark -> DarkScheme - ThemeMode.Sepia -> SepiaScheme - ThemeMode.SepiaDark -> SepiaDarkScheme - ThemeMode.Black -> BlackScheme - ThemeMode.System -> if (systemInDark) DarkScheme else LightScheme -}.background.toArgb() +fun windowBackground(mode: ThemeMode, systemInDark: Boolean, oled: Boolean): Int { + val dark = when (mode) { + ThemeMode.System -> systemInDark + ThemeMode.Light, ThemeMode.Sepia -> false + ThemeMode.Dark, ThemeMode.SepiaDark -> true + } + if (oled && dark) return Color.Black.toArgb() + return when (mode) { + ThemeMode.Light -> LightScheme + ThemeMode.Dark -> DarkScheme + ThemeMode.Sepia -> SepiaScheme + ThemeMode.SepiaDark -> SepiaDarkScheme + ThemeMode.System -> if (systemInDark) DarkScheme else LightScheme + }.background.toArgb() +} + +/** + * Pushes a dark scheme to true black for OLED panels, where an unlit pixel costs no power. + * + * Only the surfaces move, and they move towards black rather than being replaced by it, so sepia + * night keeps its warmth instead of turning into the grey dark theme. Text and accents are left + * exactly as they were. + */ +private fun ColorScheme.asOled(): ColorScheme = copy( + background = Color.Black, + surface = Color.Black, + surfaceDim = Color.Black, + surfaceContainerLowest = Color.Black, + surfaceContainerLow = lerp(surfaceContainerLow, Color.Black, 0.80f), + surfaceContainer = lerp(surfaceContainer, Color.Black, 0.74f), + surfaceContainerHigh = lerp(surfaceContainerHigh, Color.Black, 0.64f), + surfaceContainerHighest = lerp(surfaceContainerHighest, Color.Black, 0.54f), + surfaceBright = lerp(surfaceBright, Color.Black, 0.46f), + surfaceVariant = lerp(surfaceVariant, Color.Black, 0.52f), +) @Composable -fun GanjoorTheme(mode: ThemeMode, language: Language, content: @Composable () -> Unit) { +fun GanjoorTheme( + mode: ThemeMode, + language: Language, + oled: Boolean, + content: @Composable () -> Unit, +) { val dark = when (mode) { ThemeMode.System -> isSystemInDarkTheme() ThemeMode.Light, ThemeMode.Sepia -> false - ThemeMode.Dark, ThemeMode.SepiaDark, ThemeMode.Black -> true + ThemeMode.Dark, ThemeMode.SepiaDark -> true } - val scheme = when (mode) { + val chosen = when (mode) { ThemeMode.Sepia -> SepiaScheme ThemeMode.SepiaDark -> SepiaDarkScheme - ThemeMode.Black -> BlackScheme else -> if (dark) DarkScheme else LightScheme } + // Only dark schemes have anything to gain from it. + val scheme = if (oled && dark) chosen.asOled() else chosen val view = LocalView.current if (!view.isInEditMode) { diff --git a/app/src/main/res/drawable/ic_pause.xml b/app/src/main/res/drawable/ic_pause.xml new file mode 100644 index 0000000..2831bcb --- /dev/null +++ b/app/src/main/res/drawable/ic_pause.xml @@ -0,0 +1,9 @@ + + + diff --git a/app/src/main/res/values-fa/strings.xml b/app/src/main/res/values-fa/strings.xml index 2a1da44..877c051 100644 --- a/app/src/main/res/values-fa/strings.xml +++ b/app/src/main/res/values-fa/strings.xml @@ -58,11 +58,18 @@ جست‌وجو به اینترنت نیاز دارد. حالت برون‌خط را خاموش کنید. بیشتر تمام - سیاه (OLED) درباره و پروانه‌ها گنجور برای اندروید نرم‌افزار آزاد است. شعرها، قلم‌ها و همهٔ کتابخانه‌هایی که این برنامه بر آن‌ها ساخته شده در زیر آمده‌اند؛ برای خواندن متن کامل پروانه روی هر مورد بزنید. برای این واژه مدخلی یافت نشد. - ویکی‌واژه (CC BY-SA 3.0) - فرهنگ دانشجو + ویکی‌واژه — فارسی به انگلیسی (CC BY-SA 3.0) + فرهنگ دانشجو — فارسی به انگلیسی لغت‌نامه + ویکی‌واژه — اردو به انگلیسی (CC BY-SA 3.0) + ویکی‌واژهٔ اردو — اردو به اردو (CC BY-SA 3.0) + واژه‌های نزدیک + ویکی‌واژه — عربی به انگلیسی (CC BY-SA 3.0) + پس‌زمینهٔ سیاه (OLED) + روی پوسته‌های تاریک اعمال می‌شود و در نمایشگر OLED باتری کمتری می‌برد + پخش خوانش + توقف diff --git a/app/src/main/res/values-ur/strings.xml b/app/src/main/res/values-ur/strings.xml index a9dcf86..a748790 100644 --- a/app/src/main/res/values-ur/strings.xml +++ b/app/src/main/res/values-ur/strings.xml @@ -58,11 +58,18 @@ تلاش کے لیے انٹرنیٹ درکار ہے۔ آف لائن موڈ بند کریں۔ مزید مکمل - سیاہ (OLED) تعارف اور لائسنس گنجور فار اینڈرائیڈ آزاد سافٹ ویئر ہے۔ کلام، فونٹس اور تمام لائبریریاں ذیل میں درج ہیں؛ مکمل لائسنس پڑھنے کے لیے کسی اندراج پر ٹیپ کریں۔ اس لفظ کا کوئی اندراج نہیں ملا۔ - ویکی لغت (CC BY-SA 3.0) - دانشجو لغت + ویکی لغت — فارسی سے انگریزی (CC BY-SA 3.0) + دانشجو لغت — فارسی سے انگریزی لغت نامہ + ویکی لغت — اردو سے انگریزی (CC BY-SA 3.0) + اردو ویکی لغت — اردو سے اردو (CC BY-SA 3.0) + ملتے جلتے الفاظ + ویکی لغت — عربی سے انگریزی (CC BY-SA 3.0) + سیاہ پس منظر (OLED) + تاریک تھیمز پر لاگو ہوتا ہے؛ OLED اسکرین پر بیٹری بچاتا ہے + قرات سنیں + وقفہ diff --git a/app/src/main/res/values/strings.xml b/app/src/main/res/values/strings.xml index bb03491..afc5b66 100644 --- a/app/src/main/res/values/strings.xml +++ b/app/src/main/res/values/strings.xml @@ -63,11 +63,18 @@ Search needs a connection. Turn off offline mode to use it. Load more Done - Black (OLED) About & licences Ganjoor for Android is free software. The poems, the fonts and every library it is built on are credited below; tap an entry to read its full licence. No entry for this word. - Wiktionary (CC BY-SA 3.0) - Daneshjoo Dictionary + Wiktionary — Persian to English (CC BY-SA 3.0) + Daneshjoo — Persian to English Dictionary + Wiktionary — Urdu to English (CC BY-SA 3.0) + Urdu Wiktionary — Urdu to Urdu (CC BY-SA 3.0) + Similar words + Wiktionary — Arabic to English (CC BY-SA 3.0) + Black backgrounds (OLED) + Applies to the dark themes; saves power on OLED screens + Play recitation + Pause diff --git a/app/src/test/java/com/ganjoor/android/CoupletsTest.kt b/app/src/test/java/com/ganjoor/android/CoupletsTest.kt index 6ba0f3d..3f1d977 100644 --- a/app/src/test/java/com/ganjoor/android/CoupletsTest.kt +++ b/app/src/test/java/com/ganjoor/android/CoupletsTest.kt @@ -1,6 +1,10 @@ package com.ganjoor.android +import com.ganjoor.android.data.CatEntry +import com.ganjoor.android.data.Category import com.ganjoor.android.data.Crumb +import com.ganjoor.android.data.PoemRef +import com.ganjoor.android.data.orderedEntries import com.ganjoor.android.data.Verse import com.ganjoor.android.data.breadcrumbs import com.ganjoor.android.data.parentUrl @@ -125,3 +129,54 @@ class ParentUrlTest { assertEquals("/hafez", parentUrl("/hafez/ghazal/")) } } + +class CategoryOrderTest { + private fun cat(chapters: List, poems: List) = Category( + id = 1, + title = "book", + childCats = chapters.mapIndexed { i, t -> Category(id = 100 + i, title = t) }, + poems = poems.mapIndexed { i, t -> PoemRef(id = 200 + i, title = t) }, + ) + + private fun titles(category: Category) = orderedEntries(category).map { + when (it) { + is CatEntry.Chapter -> it.category.title + is CatEntry.Poem -> it.poem.title + } + } + + @Test + fun `a preface comes before the chapters, as on ganjoor net`() { + // Golestan: دیباچه then the eight باب + val golestan = cat(listOf("باب اول", "باب دوم"), listOf("دیباچه")) + + assertEquals(listOf("دیباچه", "باب اول", "باب دوم"), titles(golestan)) + } + + @Test + fun `other poems stay after the chapters`() { + // Hafez: مقدّمه, then the collections, then مثنوی and ساقی‌نامه + val hafez = cat( + chapters = listOf("غزلیات", "قطعات"), + poems = listOf("مثنوی (الا ای آهوی وحشی)", "ساقی‌نامه", "مقدّمهٔ جمع‌آورندهٔ دیوان حافظ"), + ) + + assertEquals( + listOf("مقدّمهٔ جمع‌آورندهٔ دیوان حافظ", "غزلیات", "قطعات", "مثنوی (الا ای آهوی وحشی)", "ساقی‌نامه"), + titles(hafez), + ) + } + + @Test + fun `diacritics in a preface title don't hide it`() { + // مقدّمه carries a shadda the plain spelling doesn't + assertEquals(listOf("مقدّمه", "باب اول"), titles(cat(listOf("باب اول"), listOf("مقدّمه")))) + } + + @Test + fun `a category with no poems is left exactly as it is`() { + val masnavi = cat(listOf("دفتر اول", "دفتر دوم", "دفتر سوم"), emptyList()) + + assertEquals(listOf("دفتر اول", "دفتر دوم", "دفتر سوم"), titles(masnavi)) + } +} diff --git a/app/src/test/java/com/ganjoor/android/DictionaryTest.kt b/app/src/test/java/com/ganjoor/android/DictionaryTest.kt index 0724708..8f7f09f 100644 --- a/app/src/test/java/com/ganjoor/android/DictionaryTest.kt +++ b/app/src/test/java/com/ganjoor/android/DictionaryTest.kt @@ -1,6 +1,7 @@ package com.ganjoor.android import com.ganjoor.android.data.affixes +import com.ganjoor.android.data.letterOverlap import com.ganjoor.android.data.normalise import com.ganjoor.android.data.wordAt import org.junit.Assert.assertEquals @@ -79,3 +80,58 @@ class ArabicArticleTest { assertTrue(affixes("الناس").contains("ناس")) } } + +class PersianMorphologyTest { + @Test + fun `enclitic pronouns glued onto a verb are stripped`() { + assertTrue(affixes("آیدت").contains("آید")) + assertTrue(affixes("باشدش").contains("باشد")) + assertTrue(affixes("تربتش").contains("تربت")) + } + + @Test + fun `a prefix and a negation together still reach the verb`() { + // برنیاید = بر + ن + یاید; one pass would stop at نیاید + assertTrue(affixes("برنیاید").contains("یاید")) + } + + @Test + fun `plural and object markers still work`() { + assertTrue(affixes("دلها").contains("دل")) + assertTrue(affixes("مارا").contains("ما")) + } + + @Test + fun `stripping never produces a single letter`() { + assertTrue(affixes("شان").none { it.length < 2 }) + assertTrue(affixes("بها").none { it.length < 2 }) + } +} + +class LetterOverlapTest { + @Test + fun `an identical word overlaps completely`() { + assertEquals(1f, letterOverlap("عشق", "عشق"), 0.001f) + } + + @Test + fun `a suffixed form still scores high against its stem`() { + assertTrue(letterOverlap("مشکل", "مشکلها") > 0.6f) + } + + @Test + fun `sharing only a first letter scores low`() { + assertTrue(letterOverlap("عشق", "عبادتگاه") < 0.4f) + } + + @Test + fun `letters are counted once each, not by presence alone`() { + // ااا against ا shares one letter, not three + assertEquals(1f / 3f, letterOverlap("ااا", "ا"), 0.001f) + } + + @Test + fun `an empty word never matches`() { + assertEquals(0f, letterOverlap("", "عشق"), 0.001f) + } +} diff --git a/licenses/README.md b/licenses/README.md index 1dcd69b..ab5758a 100644 --- a/licenses/README.md +++ b/licenses/README.md @@ -19,6 +19,29 @@ bundled rather than merely linked. Nothing in Ganjoor's repositories restricts AI-assisted use. +## Dictionary + +Five sources, each row in the database tagged with the one it came from so the app can name it +and so any of them can be dropped without rebuilding the others. + +| Source | Direction | Licence | +|---|---|---| +| [Wiktionary](https://en.wiktionary.org) | Persian → English | CC BY-SA 3.0 | +| [Wiktionary](https://en.wiktionary.org) | Urdu → English | CC BY-SA 3.0 | +| [Urdu Wiktionary](https://ur.wiktionary.org) | Urdu → Urdu | CC BY-SA 3.0 | +| [Wiktionary](https://en.wiktionary.org) | Arabic → English | CC BY-SA 3.0 | +| [Daneshjoo](https://github.com/0xdolan/Daneshjoo) | Persian → English | Repository states MIT | + +Pronunciation — IPA with the variety it belongs to, and Urdu Wiktionary's vowelled spelling and +syllable split — comes from the same Wiktionary exports and carries the same licence. + +Because four of the five are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**. +Attribution is shown in the app beside every definition. See [`../tools/README.md`](../tools/README.md) +to rebuild it. + +The Daneshjoo entry is worth a caveat: the repository states MIT, but the underlying lexicon is +a published Iranian dictionary, so that relicensing is worth verifying before relying on it. + ## Fonts | Font | Copyright | Licence | File | diff --git a/tools/README.md b/tools/README.md index 43f732f..8f72f80 100644 --- a/tools/README.md +++ b/tools/README.md @@ -30,6 +30,39 @@ tables, etymology templates, IPA and descendants. The build keeps the definition form→lemma index (149,589 pairs) and drops the rest. That index is what resolves conjugated verbs: `افتاد → افتادن`, `بگشاید → گشودن`, `دانند → دانستن`. +## Urdu sources + +```sh +curl -L -o ur.jsonl https://kaikki.org/dictionary/Urdu/kaikki.org-dictionary-Urdu.jsonl +curl -L -o urwikt.xml.bz2 \ + https://dumps.wikimedia.org/urwiktionary/latest/urwiktionary-latest-pages-articles.xml.bz2 +bunzip2 -k urwikt.xml.bz2 +``` + +The first gives Urdu headwords glossed in English. The second is Urdu Wiktionary itself, the only +source here whose definitions are written **in Urdu** — thin (around 3,100 usable entries out of +31,000 pages, many being stubs), but for a word it does carry an Urdu reader is better served by +it than by a translation into English. + +## Arabic + +```sh +curl -L -o ar.jsonl https://kaikki.org/dictionary/Arabic/kaikki.org-dictionary-Arabic.jsonl +``` + +521 MB, almost all of it the inflection index, and that index is the point: the Arabic quoted +inside Persian verse is conjugated, so السّاقی, الناس, تَلْقَ and تَهْوی only reach a definition +through it. 36,627 entries and 819,608 new form pairs for about 59 MB of database. + +## Why there is no Persian-to-Urdu + +Wiktionary's Persian entries carry no translations at all — the translation tables live only on +English pages, in a 3.3 GB export. Going Persian to Urdu would mean pivoting through an English +sense, and a sample of that file projects only about 6,900 Persian words with any Urdu +equivalent, most of them modern dictionary vocabulary rather than the language of the poems. The +definitions written in Urdu therefore come from Urdu Wiktionary directly, and are preferred over +the English ones wherever they exist. + ## Licences Each row carries its `source`, so attribution stays accurate and either source can be dropped diff --git a/tools/build_dictionary.py b/tools/build_dictionary.py index 2bcddab..6271063 100644 --- a/tools/build_dictionary.py +++ b/tools/build_dictionary.py @@ -1,5 +1,11 @@ """Builds the app's dictionary from Wiktionary (CC BY-SA 3.0) and Daneshjoo (MIT). +Three sources, each row tagged so the app can say which one answered and in which language: + wiktionary-fa Persian headwords, English definitions + daneshjoo Persian headwords, English definitions + wiktionary-ur Urdu headwords, English definitions + + Wiktionary's export is 93 MB of linguistic metadata around 0.94 MB of definitions, so this keeps the definitions and the form->lemma index and discards the rest. The `source` column is what keeps the attribution honest and lets either source be dropped later. @@ -23,9 +29,20 @@ c = sqlite3.connect(db) c.executescript(""" CREATE TABLE entry (word TEXT NOT NULL, display TEXT NOT NULL, gloss TEXT NOT NULL, source TEXT NOT NULL); CREATE TABLE form (form TEXT NOT NULL, lemma TEXT NOT NULL); +CREATE TABLE pron (word TEXT NOT NULL, text TEXT NOT NULL, label TEXT NOT NULL, source TEXT NOT NULL); """) -entries, forms = [], set() +entries, forms, prons = [], set(), set() + +def collect_sounds(word, entry, source): + """IPA with whatever dialect it belongs to. Classical Persian matters most here: this is an + app for poetry written a long time before modern Tehrani vowels.""" + for sound in entry.get('sounds') or []: + ipa = (sound.get('ipa') or '').strip() + if not ipa: + continue + tags = [t for t in (sound.get('tags') or []) if t not in ('formal',)] + prons.add((normalise(word), ipa, ' '.join(tags), source)) for line in open('fa.jsonl', encoding='utf-8'): try: e = json.loads(line) @@ -36,13 +53,54 @@ for line in open('fa.jsonl', encoding='utf-8'): if gs: pos = e.get('pos') or '' gloss = '; '.join(dict.fromkeys(gs))[:600] - entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary')) + entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa')) + collect_sounds(word, e, 'wiktionary-fa') for f in e.get('forms', []): t = f.get('form') if t and t != word and not t.startswith('-') and len(t) > 1: forms.add((normalise(t), normalise(word))) -print(f"wiktionary: {len(entries)} entries, {len(forms)} forms") +print(f"wiktionary-fa: {len(entries)} entries, {len(forms)} forms") + +# Urdu shares a great deal of vocabulary with Persian, so these entries answer words the +# Persian sources miss. Headwords are Urdu; the definitions are still English. +n_fa = len(entries) +for line in open('ur.jsonl', encoding='utf-8'): + try: e = json.loads(line) + except Exception: continue + word = e.get('word') + if not word: continue + gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()] + if gs: + pos = e.get('pos') or '' + gloss = '; '.join(dict.fromkeys(gs))[:600] + entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur')) + collect_sounds(word, e, 'wiktionary-ur') + for f in e.get('forms', []): + t = f.get('form') + if t and t != word and not t.startswith('-') and len(t) > 1: + forms.add((normalise(t), normalise(word))) +print(f"wiktionary-ur: {len(entries) - n_fa} entries, {len(forms)} forms total") + +# Arabic, for the lines classical Persian quotes outright — Hafez opens with one. +if os.path.exists('ar.jsonl'): + n_ar, f_ar = len(entries), len(forms) + for line in open('ar.jsonl', encoding='utf-8'): + try: e = json.loads(line) + except Exception: continue + word = e.get('word') + if not word: continue + gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()] + if gs: + pos = e.get('pos') or '' + gloss = '; '.join(dict.fromkeys(gs))[:600] + entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar')) + collect_sounds(word, e, 'wiktionary-ar') + for f in e.get('forms', []): + t = f.get('form') + if t and t != word and not t.startswith('-') and len(t) > 1: + forms.add((normalise(t), normalise(word))) + print(f"wiktionary-ar: {len(entries) - n_ar} entries, {len(forms) - f_ar} new forms") tag = re.compile(r'<[^>]+>') n0 = len(entries) @@ -58,11 +116,57 @@ for k, v in MDX('daneshjoo.mdx').items(): entries.append((normalise(word), word, txt[:600], 'daneshjoo')) print(f"daneshjoo : {len(entries) - n0} entries") +# Urdu Wiktionary, the only source here whose definitions are written in Urdu rather than +# English. Thin — a few thousand usable entries, many pages being stubs — but for a word it +# does carry, an Urdu reader is better served by it than by a translation into English. +if os.path.exists('urwikt.xml'): + import html as _html + raw = open('urwikt.xml', encoding='utf-8', errors='ignore').read() + n_ur = len(entries) + for title, ns, body in re.findall( + r'(.*?).*?(\d+).*?]*>(.*?)', raw, re.S + ): + if ns != '0' or not re.match(r'^[\u0600-\u06FF]', title): + continue + t = re.sub(r'\{\{[^}]*\}\}', ' ', body) + t = re.sub(r'\[\[([^\]|]*\|)?([^\]]*)\]\]', r'\2', t) + t = re.sub(r"'{2,}|<[^>]+>", '', _html.unescape(t)) + section = re.search(r'==\s*معانی\s*==(.*?)(?:\n==|\Z)', t, re.S) + lines = (section.group(1) if section else + '\n'.join(l.strip(' #') for l in t.split('\n') if l.strip().startswith('#'))) + kept = [] + for line in lines.split('\n'): + line = re.sub(r'^\d+\.\s*', '', line.strip()) + # ؎ introduces a verse citation, and "ref"/a year starts the source note; the + # definition itself is what comes before either. + if line.startswith('؎') or re.match(r'^\(?\s*\d{3,4}ء', line): + break + line = re.split(r'\bref\b|؎', line)[0].strip() + if not line or not re.search(r'[\u0600-\u06FF]', line): + continue + kept.append(line) + if len(' '.join(kept)) > 220: + break + gloss = ' '.join(kept).strip()[:300] + if len(gloss) > 3: + entries.append((normalise(title), title, gloss, 'urwiktionary')) + # {عِشْق} is the fully vowelled spelling; "عِش + قوں" is the syllable split + vowelled = re.search(r'\{([\u0600-\u06FF\u064B-\u0652 ]{2,40})\}', t) + if vowelled: + prons.add((normalise(title), vowelled.group(1).strip(), 'اردو', 'urwiktionary')) + syllables = re.search(r'\{([\u0600-\u06FF\u064B-\u0652]+(?: \+ [\u0600-\u06FF\u064B-\u0652]+)+[^}]*)\}', t) + if syllables: + prons.add((normalise(title), syllables.group(1).strip(), 'ہجے', 'urwiktionary')) + print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)") + +print(f"pronunciations: {len(prons)}") +c.executemany("INSERT INTO pron VALUES (?,?,?,?)", sorted(prons)) c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries) c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms)) c.executescript(""" CREATE INDEX idx_entry_word ON entry(word); CREATE INDEX idx_form_form ON form(form); +CREATE INDEX idx_pron_word ON pron(word); """) c.commit() c.execute("VACUUM")