Show how a word is pronounced, Classical Persian first
Wiktionary carries pronunciation for 45,000 of these headwords and the build was throwing it away. It is now kept: 101,306 entries of IPA, each tagged with the variety it belongs to, which matters here because a word in a 14th-century ghazal was not said the way Tehran says it now — عشق is /ˈʔiʃq/ in Classical Persian and [ʔeʃɢ̥] in Iran today, and the app leads with the former. The IPA's dots and stress marks are the syllable breakdown. Urdu Wiktionary adds its own, in Urdu script: the fully vowelled spelling عِشْق and the split عِش + قوں, both pulled out of its wikitext. Costs 7.7 MB of database, taking it to 94 MB. Verified on an API 36 emulator: tapping عشق shows the Classical Persian IPA, the Urdu syllable split, the vowelled spelling and two modern variants above the definitions. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
b16d020ce0
commit
0df736fd89
Binary file not shown.
@ -11,13 +11,19 @@ import java.text.Normalizer
|
|||||||
/** One definition, and where it came from, so the credit stays attached to the text. */
|
/** One definition, and where it came from, so the credit stays attached to the text. */
|
||||||
data class Definition(val word: String, val gloss: String, val source: String)
|
data class Definition(val word: String, val gloss: String, val source: String)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* How a word sounds. [label] is the variety it belongs to — Classical Persian, Dari, Standard
|
||||||
|
* Urdu — because a word in a 14th-century ghazal was not said the way Tehran says it now.
|
||||||
|
*/
|
||||||
|
data class Pronunciation(val text: String, val label: String)
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Word lookup over a bundled SQLite built from Wiktionary (CC BY-SA 3.0) and Daneshjoo.
|
* Word lookup over a bundled SQLite built from Wiktionary (CC BY-SA 3.0) and Daneshjoo.
|
||||||
* See `tools/build_dictionary.py`; the asset ships gzipped and is unpacked once on first use.
|
* See `tools/build_dictionary.py`; the asset ships gzipped and is unpacked once on first use.
|
||||||
*/
|
*/
|
||||||
object Dictionary {
|
object Dictionary {
|
||||||
/** Guards against a copy interrupted half-way leaving an unopenable file behind. */
|
/** Guards against a copy interrupted half-way leaving an unopenable file behind. */
|
||||||
private const val ASSET_BYTES = 86663168L
|
private const val ASSET_BYTES = 94384128L
|
||||||
|
|
||||||
// Not a .gz: the build packager silently gunzips those and drops the extension, which left
|
// Not a .gz: the build packager silently gunzips those and drops the extension, which left
|
||||||
// the asset under a different name than the code was opening.
|
// the asset under a different name than the code was opening.
|
||||||
@ -46,6 +52,37 @@ object Dictionary {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* How [raw] is pronounced, Classical Persian first: this is an app for poetry written long
|
||||||
|
* before modern Tehrani vowels, and the IPA's dots and stress marks are the syllable
|
||||||
|
* breakdown that goes with it.
|
||||||
|
*/
|
||||||
|
suspend fun pronunciations(raw: String, limit: Int = 5): List<Pronunciation> =
|
||||||
|
withContext(Dispatchers.IO) {
|
||||||
|
val database = open() ?: return@withContext emptyList()
|
||||||
|
val word = normalise(raw).takeIf { it.isNotEmpty() } ?: return@withContext emptyList()
|
||||||
|
database.rawQuery(
|
||||||
|
"""
|
||||||
|
SELECT text, label FROM pron WHERE word = ?
|
||||||
|
ORDER BY CASE
|
||||||
|
WHEN label LIKE 'Classical%' THEN 0
|
||||||
|
WHEN source = 'urwiktionary' THEN 1
|
||||||
|
WHEN source = 'wiktionary-fa' THEN 2
|
||||||
|
WHEN source = 'wiktionary-ur' THEN 3
|
||||||
|
ELSE 4
|
||||||
|
END
|
||||||
|
LIMIT ?
|
||||||
|
""",
|
||||||
|
arrayOf(word, limit.toString()),
|
||||||
|
).use { cursor ->
|
||||||
|
buildList {
|
||||||
|
while (cursor.moveToNext()) {
|
||||||
|
add(Pronunciation(cursor.getString(0), cursor.getString(1)))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Looks a word up, widening the search until something matches:
|
* Looks a word up, widening the search until something matches:
|
||||||
*
|
*
|
||||||
|
|||||||
@ -30,6 +30,7 @@ import androidx.compose.ui.unit.dp
|
|||||||
import com.ganjoor.android.R
|
import com.ganjoor.android.R
|
||||||
import com.ganjoor.android.data.Definition
|
import com.ganjoor.android.data.Definition
|
||||||
import com.ganjoor.android.data.Dictionary
|
import com.ganjoor.android.data.Dictionary
|
||||||
|
import com.ganjoor.android.data.Pronunciation
|
||||||
import com.ganjoor.android.ui.theme.readingStyle
|
import com.ganjoor.android.ui.theme.readingStyle
|
||||||
|
|
||||||
/** English prose inside an otherwise right-to-left sheet. */
|
/** English prose inside an otherwise right-to-left sheet. */
|
||||||
@ -69,6 +70,9 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
|
|||||||
val definitions by produceState<List<Definition>?>(null, current) {
|
val definitions by produceState<List<Definition>?>(null, current) {
|
||||||
value = Dictionary.lookup(current)
|
value = Dictionary.lookup(current)
|
||||||
}
|
}
|
||||||
|
val sounds by produceState(emptyList<Pronunciation>(), current) {
|
||||||
|
value = Dictionary.pronunciations(current)
|
||||||
|
}
|
||||||
val suggestions by produceState(emptyList<String>(), current, definitions) {
|
val suggestions by produceState(emptyList<String>(), current, definitions) {
|
||||||
value = if (definitions?.isEmpty() == true) Dictionary.suggest(current) else emptyList()
|
value = if (definitions?.isEmpty() == true) Dictionary.suggest(current) else emptyList()
|
||||||
}
|
}
|
||||||
@ -88,6 +92,25 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
|
|||||||
style = readingStyle(prefs.font, prefs.fontSize, prefs.fontWeight.weight),
|
style = readingStyle(prefs.font, prefs.fontSize, prefs.fontWeight.weight),
|
||||||
modifier = Modifier.fillMaxWidth(),
|
modifier = Modifier.fillMaxWidth(),
|
||||||
)
|
)
|
||||||
|
if (sounds.isNotEmpty()) {
|
||||||
|
FlowRow(horizontalArrangement = Arrangement.spacedBy(12.dp)) {
|
||||||
|
sounds.forEach { sound ->
|
||||||
|
Column {
|
||||||
|
// IPA is Latin-script, so it reads left to right whatever the UI does
|
||||||
|
LeftToRight {
|
||||||
|
Text(sound.text, style = MaterialTheme.typography.bodyMedium)
|
||||||
|
}
|
||||||
|
if (sound.label.isNotBlank()) {
|
||||||
|
Text(
|
||||||
|
text = sound.label,
|
||||||
|
style = MaterialTheme.typography.labelSmall,
|
||||||
|
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
HorizontalDivider()
|
HorizontalDivider()
|
||||||
|
|
||||||
when {
|
when {
|
||||||
|
|||||||
@ -21,7 +21,7 @@ Nothing in Ganjoor's repositories restricts AI-assisted use.
|
|||||||
|
|
||||||
## Dictionary
|
## Dictionary
|
||||||
|
|
||||||
Four sources, each row in the database tagged with the one it came from so the app can name it
|
Five sources, each row in the database tagged with the one it came from so the app can name it
|
||||||
and so any of them can be dropped without rebuilding the others.
|
and so any of them can be dropped without rebuilding the others.
|
||||||
|
|
||||||
| Source | Direction | Licence |
|
| Source | Direction | Licence |
|
||||||
@ -29,9 +29,13 @@ and so any of them can be dropped without rebuilding the others.
|
|||||||
| [Wiktionary](https://en.wiktionary.org) | Persian → English | CC BY-SA 3.0 |
|
| [Wiktionary](https://en.wiktionary.org) | Persian → English | CC BY-SA 3.0 |
|
||||||
| [Wiktionary](https://en.wiktionary.org) | Urdu → English | CC BY-SA 3.0 |
|
| [Wiktionary](https://en.wiktionary.org) | Urdu → English | CC BY-SA 3.0 |
|
||||||
| [Urdu Wiktionary](https://ur.wiktionary.org) | Urdu → Urdu | CC BY-SA 3.0 |
|
| [Urdu Wiktionary](https://ur.wiktionary.org) | Urdu → Urdu | CC BY-SA 3.0 |
|
||||||
|
| [Wiktionary](https://en.wiktionary.org) | Arabic → English | CC BY-SA 3.0 |
|
||||||
| [Daneshjoo](https://github.com/0xdolan/Daneshjoo) | Persian → English | Repository states MIT |
|
| [Daneshjoo](https://github.com/0xdolan/Daneshjoo) | Persian → English | Repository states MIT |
|
||||||
|
|
||||||
Because three of the four are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**.
|
Pronunciation — IPA with the variety it belongs to, and Urdu Wiktionary's vowelled spelling and
|
||||||
|
syllable split — comes from the same Wiktionary exports and carries the same licence.
|
||||||
|
|
||||||
|
Because four of the five are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**.
|
||||||
Attribution is shown in the app beside every definition. See [`../tools/README.md`](../tools/README.md)
|
Attribution is shown in the app beside every definition. See [`../tools/README.md`](../tools/README.md)
|
||||||
to rebuild it.
|
to rebuild it.
|
||||||
|
|
||||||
|
|||||||
@ -29,9 +29,20 @@ c = sqlite3.connect(db)
|
|||||||
c.executescript("""
|
c.executescript("""
|
||||||
CREATE TABLE entry (word TEXT NOT NULL, display TEXT NOT NULL, gloss TEXT NOT NULL, source TEXT NOT NULL);
|
CREATE TABLE entry (word TEXT NOT NULL, display TEXT NOT NULL, gloss TEXT NOT NULL, source TEXT NOT NULL);
|
||||||
CREATE TABLE form (form TEXT NOT NULL, lemma TEXT NOT NULL);
|
CREATE TABLE form (form TEXT NOT NULL, lemma TEXT NOT NULL);
|
||||||
|
CREATE TABLE pron (word TEXT NOT NULL, text TEXT NOT NULL, label TEXT NOT NULL, source TEXT NOT NULL);
|
||||||
""")
|
""")
|
||||||
|
|
||||||
entries, forms = [], set()
|
entries, forms, prons = [], set(), set()
|
||||||
|
|
||||||
|
def collect_sounds(word, entry, source):
|
||||||
|
"""IPA with whatever dialect it belongs to. Classical Persian matters most here: this is an
|
||||||
|
app for poetry written a long time before modern Tehrani vowels."""
|
||||||
|
for sound in entry.get('sounds') or []:
|
||||||
|
ipa = (sound.get('ipa') or '').strip()
|
||||||
|
if not ipa:
|
||||||
|
continue
|
||||||
|
tags = [t for t in (sound.get('tags') or []) if t not in ('formal',)]
|
||||||
|
prons.add((normalise(word), ipa, ' '.join(tags), source))
|
||||||
|
|
||||||
for line in open('fa.jsonl', encoding='utf-8'):
|
for line in open('fa.jsonl', encoding='utf-8'):
|
||||||
try: e = json.loads(line)
|
try: e = json.loads(line)
|
||||||
@ -43,6 +54,7 @@ for line in open('fa.jsonl', encoding='utf-8'):
|
|||||||
pos = e.get('pos') or ''
|
pos = e.get('pos') or ''
|
||||||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||||||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa'))
|
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa'))
|
||||||
|
collect_sounds(word, e, 'wiktionary-fa')
|
||||||
for f in e.get('forms', []):
|
for f in e.get('forms', []):
|
||||||
t = f.get('form')
|
t = f.get('form')
|
||||||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||||||
@ -63,6 +75,7 @@ for line in open('ur.jsonl', encoding='utf-8'):
|
|||||||
pos = e.get('pos') or ''
|
pos = e.get('pos') or ''
|
||||||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||||||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur'))
|
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur'))
|
||||||
|
collect_sounds(word, e, 'wiktionary-ur')
|
||||||
for f in e.get('forms', []):
|
for f in e.get('forms', []):
|
||||||
t = f.get('form')
|
t = f.get('form')
|
||||||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||||||
@ -82,6 +95,7 @@ if os.path.exists('ar.jsonl'):
|
|||||||
pos = e.get('pos') or ''
|
pos = e.get('pos') or ''
|
||||||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||||||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar'))
|
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ar'))
|
||||||
|
collect_sounds(word, e, 'wiktionary-ar')
|
||||||
for f in e.get('forms', []):
|
for f in e.get('forms', []):
|
||||||
t = f.get('form')
|
t = f.get('form')
|
||||||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||||||
@ -136,13 +150,23 @@ if os.path.exists('urwikt.xml'):
|
|||||||
gloss = ' '.join(kept).strip()[:300]
|
gloss = ' '.join(kept).strip()[:300]
|
||||||
if len(gloss) > 3:
|
if len(gloss) > 3:
|
||||||
entries.append((normalise(title), title, gloss, 'urwiktionary'))
|
entries.append((normalise(title), title, gloss, 'urwiktionary'))
|
||||||
|
# {عِشْق} is the fully vowelled spelling; "عِش + قوں" is the syllable split
|
||||||
|
vowelled = re.search(r'\{([\u0600-\u06FF\u064B-\u0652 ]{2,40})\}', t)
|
||||||
|
if vowelled:
|
||||||
|
prons.add((normalise(title), vowelled.group(1).strip(), 'اردو', 'urwiktionary'))
|
||||||
|
syllables = re.search(r'\{([\u0600-\u06FF\u064B-\u0652]+(?: \+ [\u0600-\u06FF\u064B-\u0652]+)+[^}]*)\}', t)
|
||||||
|
if syllables:
|
||||||
|
prons.add((normalise(title), syllables.group(1).strip(), 'ہجے', 'urwiktionary'))
|
||||||
print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)")
|
print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)")
|
||||||
|
|
||||||
|
print(f"pronunciations: {len(prons)}")
|
||||||
|
c.executemany("INSERT INTO pron VALUES (?,?,?,?)", sorted(prons))
|
||||||
c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries)
|
c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries)
|
||||||
c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms))
|
c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms))
|
||||||
c.executescript("""
|
c.executescript("""
|
||||||
CREATE INDEX idx_entry_word ON entry(word);
|
CREATE INDEX idx_entry_word ON entry(word);
|
||||||
CREATE INDEX idx_form_form ON form(form);
|
CREATE INDEX idx_form_form ON form(form);
|
||||||
|
CREATE INDEX idx_pron_word ON pron(word);
|
||||||
""")
|
""")
|
||||||
c.commit()
|
c.commit()
|
||||||
c.execute("VACUUM")
|
c.execute("VACUUM")
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user