Add the Urdu dictionaries, and suggest near words when nothing matches
Two Urdu sources join the Persian ones. Wiktionary's Urdu extract gives Urdu headwords glossed in English, and Urdu Wiktionary itself gives definitions written in Urdu — the only source here that does. The latter is thin, about 3,100 usable entries out of 31,000 pages since many are stubs, but for a word it carries an Urdu reader is better served by it than by a translation into English: عشق comes back as شدید جذبۂ محبت، گہری چاہت، محبت، پریم، پیار. Every definition now names the dictionary and its language pair, and lays out in the direction its own script reads, so an Urdu definition is right-aligned beside a left-aligned English one. When nothing matches, the sheet offers near words ranked by how many letters they share with what was looked up, drawn from an index range scan on the leading letters rather than a scan of the whole table. خودکامی, which has no entry, offers خودکامه — the lemma it wants. Measured honestly: the Urdu sources add little coverage over Persian — nine words from the English-glossed extract, two from Urdu Wiktionary, against 1,285 from thirteen poems. They are here because an Urdu reader wants Urdu, not because they widen the net. The suggestion ranking is verified against the built database rather than only on device: خودکامی → خودکامه, شیرازی → شیراز, مشکلها → مشکل. On an API 36 emulator all four sources answer عشق with their labels. Known rough edge: affix stripping across four languages can mislead. ناولها reaches ناول, the Urdu for "novel", which is not what Hafez meant. The matched headword is always shown, so it is visible rather than silent. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
8aad086b08
commit
c0cb6a8ab6
Binary file not shown.
@ -17,7 +17,7 @@ data class Definition(val word: String, val gloss: String, val source: String)
|
||||
*/
|
||||
object Dictionary {
|
||||
/** Guards against a copy interrupted half-way leaving an unopenable file behind. */
|
||||
private const val ASSET_BYTES = 22_298_624L
|
||||
private const val ASSET_BYTES = 27742208L
|
||||
|
||||
// Not a .gz: the build packager silently gunzips those and drops the extension, which left
|
||||
// the asset under a different name than the code was opening.
|
||||
@ -95,6 +95,47 @@ object Dictionary {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Headwords that look like [raw], for when nothing matched exactly — a misread letter, an
|
||||
* unusual spelling, or a word the dictionary simply spells differently.
|
||||
*
|
||||
* Candidates come from an index range scan on the first letters, then are ranked by how
|
||||
* many letters they share with the query. ponytail: shared letters rather than an edit
|
||||
* distance, which would need the whole table scanned to be worth the extra precision.
|
||||
*/
|
||||
suspend fun suggest(raw: String, limit: Int = 6): List<String> = withContext(Dispatchers.IO) {
|
||||
val database = open() ?: return@withContext emptyList()
|
||||
val word = normalise(raw).takeIf { it.length > 1 } ?: return@withContext emptyList()
|
||||
|
||||
// Widen the prefix until there is something to rank, but never scan the whole table.
|
||||
val candidates = generateSequence(minOf(3, word.length - 1)) { (it - 1).takeIf { n -> n >= 1 } }
|
||||
.map { prefixLength -> byPrefix(database, word.take(prefixLength)) }
|
||||
.firstOrNull { it.size >= 3 }
|
||||
?: return@withContext emptyList()
|
||||
|
||||
candidates
|
||||
.asSequence()
|
||||
.filter { it.first != word }
|
||||
.map { (normalised, display) -> display to letterOverlap(word, normalised) }
|
||||
.filter { it.second > 0.45f }
|
||||
.sortedByDescending { it.second }
|
||||
.map { it.first }
|
||||
.distinct()
|
||||
.take(limit)
|
||||
.toList()
|
||||
}
|
||||
|
||||
/** Index range scan: everything whose normalised form starts with [prefix]. */
|
||||
private fun byPrefix(database: SQLiteDatabase, prefix: String): List<Pair<String, String>> =
|
||||
database.rawQuery(
|
||||
"SELECT DISTINCT word, display FROM entry WHERE word >= ? AND word < ? LIMIT 400",
|
||||
arrayOf(prefix, prefix + '\uFFFF'),
|
||||
).use { cursor ->
|
||||
buildList {
|
||||
while (cursor.moveToNext()) add(cursor.getString(0) to cursor.getString(1))
|
||||
}
|
||||
}
|
||||
|
||||
private fun lemmas(database: SQLiteDatabase, form: String): List<String> =
|
||||
database.rawQuery(
|
||||
"SELECT lemma FROM form WHERE form = ? LIMIT 6",
|
||||
@ -152,6 +193,17 @@ internal fun normalise(text: String, keepZwnj: Boolean = false): String {
|
||||
return (if (keepZwnj) folded else folded.replace(ZWNJ.toString(), "")).trim()
|
||||
}
|
||||
|
||||
/**
|
||||
* How much of two words' letters coincide — the overlapping letters counted against the longer
|
||||
* word, so مشکل and مشکلها score high while a word that merely starts the same does not.
|
||||
*/
|
||||
internal fun letterOverlap(a: String, b: String): Float {
|
||||
if (a.isEmpty() || b.isEmpty()) return 0f
|
||||
val remaining = b.toMutableList()
|
||||
val shared = a.count { remaining.remove(it) }
|
||||
return shared.toFloat() / maxOf(a.length, b.length)
|
||||
}
|
||||
|
||||
/** The whole word surrounding [index], for turning a tap into something to look up. */
|
||||
internal fun wordAt(text: String, index: Int): String? {
|
||||
if (text.isEmpty()) return null
|
||||
|
||||
@ -64,11 +64,17 @@ private val CREDITS = listOf(
|
||||
),
|
||||
Credit(
|
||||
"Wiktionary",
|
||||
"Wiktionary contributors — the word definitions, and the inflected-form index that " +
|
||||
"finds a conjugated verb's dictionary entry",
|
||||
"Wiktionary contributors — Persian and Urdu definitions, and the inflected-form " +
|
||||
"index that finds a conjugated verb's dictionary entry",
|
||||
"CC BY-SA 3.0. The bundled dictionary is therefore also CC BY-SA 3.0.",
|
||||
url = "https://en.wiktionary.org",
|
||||
),
|
||||
Credit(
|
||||
"Urdu Wiktionary",
|
||||
"Urdu Wiktionary contributors — the definitions written in Urdu rather than English",
|
||||
"CC BY-SA 3.0. The bundled dictionary is therefore also CC BY-SA 3.0.",
|
||||
url = "https://ur.wiktionary.org",
|
||||
),
|
||||
Credit(
|
||||
"Daneshjoo Dictionary",
|
||||
"Layered under Wiktionary for the words it doesn't carry",
|
||||
|
||||
@ -7,17 +7,21 @@ import androidx.compose.foundation.layout.navigationBarsPadding
|
||||
import androidx.compose.foundation.layout.padding
|
||||
import androidx.compose.foundation.rememberScrollState
|
||||
import androidx.compose.foundation.verticalScroll
|
||||
import androidx.compose.foundation.layout.FlowRow
|
||||
import androidx.compose.material3.CircularProgressIndicator
|
||||
import androidx.compose.material3.ExperimentalMaterial3Api
|
||||
import androidx.compose.material3.HorizontalDivider
|
||||
import androidx.compose.material3.MaterialTheme
|
||||
import androidx.compose.material3.ModalBottomSheet
|
||||
import androidx.compose.material3.SuggestionChip
|
||||
import androidx.compose.material3.Text
|
||||
import androidx.compose.runtime.Composable
|
||||
import androidx.compose.runtime.CompositionLocalProvider
|
||||
import androidx.compose.runtime.getValue
|
||||
import androidx.compose.runtime.mutableStateOf
|
||||
import androidx.compose.runtime.produceState
|
||||
import androidx.compose.runtime.remember
|
||||
import androidx.compose.runtime.setValue
|
||||
import androidx.compose.ui.Modifier
|
||||
import androidx.compose.ui.platform.LocalLayoutDirection
|
||||
import androidx.compose.ui.res.stringResource
|
||||
@ -34,12 +38,39 @@ private fun LeftToRight(content: @Composable () -> Unit) {
|
||||
CompositionLocalProvider(LocalLayoutDirection provides LayoutDirection.Ltr, content = content)
|
||||
}
|
||||
|
||||
/** Lays a definition out the way its own script reads. */
|
||||
@Composable
|
||||
private fun InDirectionOf(text: String, content: @Composable () -> Unit) {
|
||||
val arabicScript = text.count { it in '\u0600'..'\u06FF' }
|
||||
val latin = text.count { it in 'A'..'Z' || it in 'a'..'z' }
|
||||
CompositionLocalProvider(
|
||||
LocalLayoutDirection provides
|
||||
if (arabicScript > latin) LayoutDirection.Rtl else LayoutDirection.Ltr,
|
||||
content = content,
|
||||
)
|
||||
}
|
||||
|
||||
/** Which dictionary answered, and in which language pair. */
|
||||
private fun sourceLabel(source: String) = when (source) {
|
||||
"wiktionary-fa" -> R.string.source_wiktionary
|
||||
"wiktionary-ur" -> R.string.source_wiktionary_ur
|
||||
"urwiktionary" -> R.string.source_urwiktionary
|
||||
else -> R.string.source_daneshjoo
|
||||
}
|
||||
|
||||
/** What the dictionary knows about a tapped word. */
|
||||
@OptIn(ExperimentalMaterial3Api::class)
|
||||
@Composable
|
||||
fun WordSheet(word: String, onDismiss: () -> Unit) {
|
||||
val prefs = LocalSettings.current.value
|
||||
val definitions by produceState<List<Definition>?>(null, word) { value = Dictionary.lookup(word) }
|
||||
// A suggestion replaces what is being looked up, so the sheet can be followed like a trail.
|
||||
var current by remember(word) { mutableStateOf(word) }
|
||||
val definitions by produceState<List<Definition>?>(null, current) {
|
||||
value = Dictionary.lookup(current)
|
||||
}
|
||||
val suggestions by produceState(emptyList<String>(), current, definitions) {
|
||||
value = if (definitions?.isEmpty() == true) Dictionary.suggest(current) else emptyList()
|
||||
}
|
||||
|
||||
ModalBottomSheet(onDismissRequest = onDismiss) {
|
||||
Column(
|
||||
@ -52,7 +83,7 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
|
||||
) {
|
||||
// The headword in the reading font, at reading size: it is a line of poetry, after all.
|
||||
Text(
|
||||
text = word,
|
||||
text = current,
|
||||
style = readingStyle(prefs.font, prefs.fontSize, prefs.fontWeight.weight),
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
)
|
||||
@ -61,7 +92,10 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
|
||||
when {
|
||||
definitions == null -> CircularProgressIndicator(Modifier.padding(vertical = 16.dp))
|
||||
|
||||
definitions!!.isEmpty() -> LeftToRight {
|
||||
definitions!!.isEmpty() -> Column(
|
||||
verticalArrangement = Arrangement.spacedBy(8.dp),
|
||||
) {
|
||||
LeftToRight {
|
||||
Text(
|
||||
text = stringResource(R.string.no_definition),
|
||||
style = MaterialTheme.typography.bodyMedium,
|
||||
@ -69,30 +103,41 @@ fun WordSheet(word: String, onDismiss: () -> Unit) {
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
)
|
||||
}
|
||||
if (suggestions.isNotEmpty()) {
|
||||
Text(
|
||||
text = stringResource(R.string.did_you_mean),
|
||||
style = MaterialTheme.typography.titleSmall,
|
||||
color = MaterialTheme.colorScheme.primary,
|
||||
)
|
||||
FlowRow(horizontalArrangement = Arrangement.spacedBy(8.dp)) {
|
||||
suggestions.forEach { suggestion ->
|
||||
SuggestionChip(
|
||||
onClick = { current = suggestion },
|
||||
label = { Text(suggestion) },
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
else -> definitions!!.forEach { definition ->
|
||||
Column(Modifier.fillMaxWidth()) {
|
||||
// The headword actually matched, which may be the lemma rather than the
|
||||
// word as it appears in the line. Persian, so it stays right-to-left.
|
||||
if (definition.word != word) {
|
||||
if (definition.word != current) {
|
||||
Text(
|
||||
text = definition.word,
|
||||
style = MaterialTheme.typography.titleSmall,
|
||||
color = MaterialTheme.colorScheme.primary,
|
||||
)
|
||||
}
|
||||
// The definitions are English; right-aligning them reads badly.
|
||||
LeftToRight {
|
||||
// Most definitions are English and right-aligning them reads badly;
|
||||
// the Urdu ones are right-to-left like the rest of the app.
|
||||
InDirectionOf(definition.gloss) {
|
||||
Column(Modifier.fillMaxWidth()) {
|
||||
Text(definition.gloss, style = MaterialTheme.typography.bodyMedium)
|
||||
Text(
|
||||
text = stringResource(
|
||||
if (definition.source == "wiktionary") {
|
||||
R.string.source_wiktionary
|
||||
} else {
|
||||
R.string.source_daneshjoo
|
||||
}
|
||||
),
|
||||
text = stringResource(sourceLabel(definition.source)),
|
||||
style = MaterialTheme.typography.labelSmall,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
|
||||
@ -65,4 +65,7 @@
|
||||
<string name="source_wiktionary">ویکیواژه — فارسی به انگلیسی (CC BY-SA 3.0)</string>
|
||||
<string name="source_daneshjoo">فرهنگ دانشجو — فارسی به انگلیسی</string>
|
||||
<string name="dictionary">لغتنامه</string>
|
||||
<string name="source_wiktionary_ur">ویکیواژه — اردو به انگلیسی (CC BY-SA 3.0)</string>
|
||||
<string name="source_urwiktionary">ویکیواژهٔ اردو — اردو به اردو (CC BY-SA 3.0)</string>
|
||||
<string name="did_you_mean">واژههای نزدیک</string>
|
||||
</resources>
|
||||
|
||||
@ -65,4 +65,7 @@
|
||||
<string name="source_wiktionary">ویکی لغت — فارسی سے انگریزی (CC BY-SA 3.0)</string>
|
||||
<string name="source_daneshjoo">دانشجو لغت — فارسی سے انگریزی</string>
|
||||
<string name="dictionary">لغت نامہ</string>
|
||||
<string name="source_wiktionary_ur">ویکی لغت — اردو سے انگریزی (CC BY-SA 3.0)</string>
|
||||
<string name="source_urwiktionary">اردو ویکی لغت — اردو سے اردو (CC BY-SA 3.0)</string>
|
||||
<string name="did_you_mean">ملتے جلتے الفاظ</string>
|
||||
</resources>
|
||||
|
||||
@ -70,4 +70,7 @@
|
||||
<string name="source_wiktionary">Wiktionary — Persian to English (CC BY-SA 3.0)</string>
|
||||
<string name="source_daneshjoo">Daneshjoo — Persian to English</string>
|
||||
<string name="dictionary">Dictionary</string>
|
||||
<string name="source_wiktionary_ur">Wiktionary — Urdu to English (CC BY-SA 3.0)</string>
|
||||
<string name="source_urwiktionary">Urdu Wiktionary — Urdu to Urdu (CC BY-SA 3.0)</string>
|
||||
<string name="did_you_mean">Similar words</string>
|
||||
</resources>
|
||||
|
||||
@ -1,6 +1,7 @@
|
||||
package com.ganjoor.android
|
||||
|
||||
import com.ganjoor.android.data.affixes
|
||||
import com.ganjoor.android.data.letterOverlap
|
||||
import com.ganjoor.android.data.normalise
|
||||
import com.ganjoor.android.data.wordAt
|
||||
import org.junit.Assert.assertEquals
|
||||
@ -106,3 +107,31 @@ class PersianMorphologyTest {
|
||||
assertTrue(affixes("بها").none { it.length < 2 })
|
||||
}
|
||||
}
|
||||
|
||||
class LetterOverlapTest {
|
||||
@Test
|
||||
fun `an identical word overlaps completely`() {
|
||||
assertEquals(1f, letterOverlap("عشق", "عشق"), 0.001f)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `a suffixed form still scores high against its stem`() {
|
||||
assertTrue(letterOverlap("مشکل", "مشکلها") > 0.6f)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `sharing only a first letter scores low`() {
|
||||
assertTrue(letterOverlap("عشق", "عبادتگاه") < 0.4f)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `letters are counted once each, not by presence alone`() {
|
||||
// ااا against ا shares one letter, not three
|
||||
assertEquals(1f / 3f, letterOverlap("ااا", "ا"), 0.001f)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `an empty word never matches`() {
|
||||
assertEquals(0f, letterOverlap("", "عشق"), 0.001f)
|
||||
}
|
||||
}
|
||||
|
||||
@ -19,6 +19,25 @@ bundled rather than merely linked.
|
||||
|
||||
Nothing in Ganjoor's repositories restricts AI-assisted use.
|
||||
|
||||
## Dictionary
|
||||
|
||||
Four sources, each row in the database tagged with the one it came from so the app can name it
|
||||
and so any of them can be dropped without rebuilding the others.
|
||||
|
||||
| Source | Direction | Licence |
|
||||
|---|---|---|
|
||||
| [Wiktionary](https://en.wiktionary.org) | Persian → English | CC BY-SA 3.0 |
|
||||
| [Wiktionary](https://en.wiktionary.org) | Urdu → English | CC BY-SA 3.0 |
|
||||
| [Urdu Wiktionary](https://ur.wiktionary.org) | Urdu → Urdu | CC BY-SA 3.0 |
|
||||
| [Daneshjoo](https://github.com/0xdolan/Daneshjoo) | Persian → English | Repository states MIT |
|
||||
|
||||
Because three of the four are CC BY-SA 3.0, **the generated `dictionary.db` is CC BY-SA 3.0**.
|
||||
Attribution is shown in the app beside every definition. See [`../tools/README.md`](../tools/README.md)
|
||||
to rebuild it.
|
||||
|
||||
The Daneshjoo entry is worth a caveat: the repository states MIT, but the underlying lexicon is
|
||||
a published Iranian dictionary, so that relicensing is worth verifying before relying on it.
|
||||
|
||||
## Fonts
|
||||
|
||||
| Font | Copyright | Licence | File |
|
||||
|
||||
@ -30,6 +30,20 @@ tables, etymology templates, IPA and descendants. The build keeps the definition
|
||||
form→lemma index (149,589 pairs) and drops the rest. That index is what resolves conjugated
|
||||
verbs: `افتاد → افتادن`, `بگشاید → گشودن`, `دانند → دانستن`.
|
||||
|
||||
## Urdu sources
|
||||
|
||||
```sh
|
||||
curl -L -o ur.jsonl https://kaikki.org/dictionary/Urdu/kaikki.org-dictionary-Urdu.jsonl
|
||||
curl -L -o urwikt.xml.bz2 \
|
||||
https://dumps.wikimedia.org/urwiktionary/latest/urwiktionary-latest-pages-articles.xml.bz2
|
||||
bunzip2 -k urwikt.xml.bz2
|
||||
```
|
||||
|
||||
The first gives Urdu headwords glossed in English. The second is Urdu Wiktionary itself, the only
|
||||
source here whose definitions are written **in Urdu** — thin (around 3,100 usable entries out of
|
||||
31,000 pages, many being stubs), but for a word it does carry an Urdu reader is better served by
|
||||
it than by a translation into English.
|
||||
|
||||
## Licences
|
||||
|
||||
Each row carries its `source`, so attribution stays accurate and either source can be dropped
|
||||
|
||||
@ -1,5 +1,11 @@
|
||||
"""Builds the app's dictionary from Wiktionary (CC BY-SA 3.0) and Daneshjoo (MIT).
|
||||
|
||||
Three sources, each row tagged so the app can say which one answered and in which language:
|
||||
wiktionary-fa Persian headwords, English definitions
|
||||
daneshjoo Persian headwords, English definitions
|
||||
wiktionary-ur Urdu headwords, English definitions
|
||||
|
||||
|
||||
Wiktionary's export is 93 MB of linguistic metadata around 0.94 MB of definitions, so this keeps
|
||||
the definitions and the form->lemma index and discards the rest. The `source` column is what
|
||||
keeps the attribution honest and lets either source be dropped later.
|
||||
@ -36,13 +42,32 @@ for line in open('fa.jsonl', encoding='utf-8'):
|
||||
if gs:
|
||||
pos = e.get('pos') or ''
|
||||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary'))
|
||||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-fa'))
|
||||
for f in e.get('forms', []):
|
||||
t = f.get('form')
|
||||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||||
forms.add((normalise(t), normalise(word)))
|
||||
|
||||
print(f"wiktionary: {len(entries)} entries, {len(forms)} forms")
|
||||
print(f"wiktionary-fa: {len(entries)} entries, {len(forms)} forms")
|
||||
|
||||
# Urdu shares a great deal of vocabulary with Persian, so these entries answer words the
|
||||
# Persian sources miss. Headwords are Urdu; the definitions are still English.
|
||||
n_fa = len(entries)
|
||||
for line in open('ur.jsonl', encoding='utf-8'):
|
||||
try: e = json.loads(line)
|
||||
except Exception: continue
|
||||
word = e.get('word')
|
||||
if not word: continue
|
||||
gs = [g.strip() for s in e.get('senses', []) for g in (s.get('glosses') or []) if g.strip()]
|
||||
if gs:
|
||||
pos = e.get('pos') or ''
|
||||
gloss = '; '.join(dict.fromkeys(gs))[:600]
|
||||
entries.append((normalise(word), word, f"({pos}) {gloss}" if pos else gloss, 'wiktionary-ur'))
|
||||
for f in e.get('forms', []):
|
||||
t = f.get('form')
|
||||
if t and t != word and not t.startswith('-') and len(t) > 1:
|
||||
forms.add((normalise(t), normalise(word)))
|
||||
print(f"wiktionary-ur: {len(entries) - n_fa} entries, {len(forms)} forms total")
|
||||
|
||||
tag = re.compile(r'<[^>]+>')
|
||||
n0 = len(entries)
|
||||
@ -58,6 +83,42 @@ for k, v in MDX('daneshjoo.mdx').items():
|
||||
entries.append((normalise(word), word, txt[:600], 'daneshjoo'))
|
||||
print(f"daneshjoo : {len(entries) - n0} entries")
|
||||
|
||||
# Urdu Wiktionary, the only source here whose definitions are written in Urdu rather than
|
||||
# English. Thin — a few thousand usable entries, many pages being stubs — but for a word it
|
||||
# does carry, an Urdu reader is better served by it than by a translation into English.
|
||||
if os.path.exists('urwikt.xml'):
|
||||
import html as _html
|
||||
raw = open('urwikt.xml', encoding='utf-8', errors='ignore').read()
|
||||
n_ur = len(entries)
|
||||
for title, ns, body in re.findall(
|
||||
r'<title>(.*?)</title>.*?<ns>(\d+)</ns>.*?<text[^>]*>(.*?)</text>', raw, re.S
|
||||
):
|
||||
if ns != '0' or not re.match(r'^[\u0600-\u06FF]', title):
|
||||
continue
|
||||
t = re.sub(r'\{\{[^}]*\}\}', ' ', body)
|
||||
t = re.sub(r'\[\[([^\]|]*\|)?([^\]]*)\]\]', r'\2', t)
|
||||
t = re.sub(r"'{2,}|<[^>]+>", '', _html.unescape(t))
|
||||
section = re.search(r'==\s*معانی\s*==(.*?)(?:\n==|\Z)', t, re.S)
|
||||
lines = (section.group(1) if section else
|
||||
'\n'.join(l.strip(' #') for l in t.split('\n') if l.strip().startswith('#')))
|
||||
kept = []
|
||||
for line in lines.split('\n'):
|
||||
line = re.sub(r'^\d+\.\s*', '', line.strip())
|
||||
# ؎ introduces a verse citation, and "ref"/a year starts the source note; the
|
||||
# definition itself is what comes before either.
|
||||
if line.startswith('؎') or re.match(r'^\(?\s*\d{3,4}ء', line):
|
||||
break
|
||||
line = re.split(r'\bref\b|؎', line)[0].strip()
|
||||
if not line or not re.search(r'[\u0600-\u06FF]', line):
|
||||
continue
|
||||
kept.append(line)
|
||||
if len(' '.join(kept)) > 220:
|
||||
break
|
||||
gloss = ' '.join(kept).strip()[:300]
|
||||
if len(gloss) > 3:
|
||||
entries.append((normalise(title), title, gloss, 'urwiktionary'))
|
||||
print(f"urwiktionary : {len(entries) - n_ur} entries (definitions in Urdu)")
|
||||
|
||||
c.executemany("INSERT INTO entry VALUES (?,?,?,?)", entries)
|
||||
c.executemany("INSERT INTO form VALUES (?,?)", sorted(forms))
|
||||
c.executescript("""
|
||||
|
||||
Loading…
Reference in New Issue
Block a user