diff --git a/GanjooRazor/Pages/SemanticSearch.cshtml b/GanjooRazor/Pages/SemanticSearch.cshtml index d0356b05..c024a72d 100644 --- a/GanjooRazor/Pages/SemanticSearch.cshtml +++ b/GanjooRazor/Pages/SemanticSearch.cshtml @@ -67,9 +67,16 @@ }, error: function (xhr) { status.style.display = "block"; - status.textContent = xhr.status === 503 - ? "در حال حاضر جست‌وجوی معنایی در دسترس نیست. لطفاً کمی بعد دوباره امتحان کنید." - : "خطایی رخ داد. لطفاً دوباره تلاش کنید."; + if (xhr.status === 503) { + status.textContent = "در حال حاضر جست‌وجوی معنایی در دسترس نیست. لطفاً کمی بعد دوباره امتحان کنید."; + } else if (xhr.responseText) { + // show the real server error text rather than a generic message - + // whatever RMuseum actually said, so a real failure is diagnosable + // from what's on screen instead of needing another round-trip + status.textContent = "خطا: " + xhr.responseText; + } else { + status.textContent = "خطایی رخ داد (کد " + xhr.status + "). لطفاً دوباره تلاش کنید."; + } } }); } diff --git a/RMuseum/RMuseum.xml b/RMuseum/RMuseum.xml index 369e0806..d703f23c 100644 --- a/RMuseum/RMuseum.xml +++ b/RMuseum/RMuseum.xml @@ -11377,12 +11377,43 @@ how many results to return; a sane default is applied server-side if omitted/invalid + + + Optional explicit scope, for a future UI (e.g. a poet picker) that wants to restrict + search without relying on auto-detection from the query text. Auto-detection (see + SemanticSearchService.DetectQueryScope) still runs even when these are set — an + explicit PoetId/CatId narrows further, it doesn't replace detection. + + + + + matches GanjoorVerse.VersePosition.ToString() (e.g. "Right"/"Left") - same convention + already used by the main ganjoor-data export, so any existing hemistich-pairing + frontend code (see mini-ganjoor's renderVerses) can be reused as-is. + + + + + A short preview from the poem's actual text - the first couple of couplets, in + original verse order. Not the whole poem; just enough for a result card. + + cosine similarity, 0..1 for these normalized vectors (in practice results cluster in a narrower band - this is a relative ranking signal, not a calibrated probability) + + + If the query text was recognized as referring to a specific poet and/or + category/book (e.g. "در کدام شعر حافظ" -> حافظ, "در کدام بخش شاهنامه" -> شاهنامه), + search was restricted to that scope and these are populated so the UI can show the + user what was detected ("نتایج محدود به: حافظ") rather than silently filtering. + Null/null if nothing was detected (or an explicit PoetId/CatId narrowed things + without any name being recognized in the free text). + + Updating related sections logs @@ -23104,8 +23135,24 @@ being constructed, taking down every endpoint under /api/ganjoor with a 503, not just semantic search. This class must never let a resource-loading failure become an unhandled exception that propagates past SearchAsync — see the catch below. + + LazyQueryScopeIndex is a SEPARATE lazy singleton from LazySemanticSearchResources on + purpose — poet/category name detection ("در کدام شعر حافظ") is a genuinely independent + concern from the embedding/model loading, with its own independent failure mode; a bug in + one must not be able to disable the other. + + + Walks the whole category subtree rooted at rootCatId (breadth-first, level by level) + and returns every category id in it, including rootCatId itself. Needed because a + detected "book" category (شاهنامه, غزلیات, ...) is rarely where poems live directly — + it's typically broken into many nested subcategories, with the actual poems several + levels deeper. One query per depth level, not per node — a large book with many + subcategories still only costs as many round-trips as the tree is deep, not how many + nodes it has. + + ganjoor.net banners service @@ -24462,12 +24509,39 @@ memory, not something to redo per-request. - + + + Returns the topK poem ids most similar to queryVector, ranked descending by cosine + similarity. queryVector must already be the SAME dimension as this index and, for the + score to mean what it claims (a true cosine similarity), should already be + L2-normalized the same way the indexed vectors are — see QueryEmbedder. + + If allowedPoemIds is provided (non-null), only poems in that set are eligible — + everything else is scored as excluded and can never appear in the results, however + similar it might be. Used for scoped search ("در کدام شعر حافظ" -> restrict to + Hafez's poems) — still a full scan either way, just with cheap early-outs for + excluded rows, since this corpus is small enough that a full scan is fast regardless. + + + + + Same "load once, never throw" pattern as LazySemanticSearchResources, but deliberately a + SEPARATE singleton with its own independent failure domain: if the poet/category name + lookup fails to load for any reason, that must only disable scope auto-detection + ("در کدام شعر حافظ" -> unscoped, searches everything) — it must never affect the embedding + index/model loading or plain (unscoped) search, which are a completely different concern. + + Uses a SemaphoreSlim rather than a plain `lock`, since the actual load is async (DB + queries) and you can't `await` inside a `lock` block. + + + - Returns the topK poem ids most similar to queryVector, ranked descending by cosine - similarity. queryVector must already be the SAME dimension as this index and, for the - score to mean what it claims (a true cosine similarity), should already be - L2-normalized the same way the indexed vectors are — see QueryEmbedder. + Loads (once) and returns the scope index, or null if loading failed or hasn't + succeeded yet — NEVER throws. The passed-in context is only actually used on the + first call that does real work; a context created by the caller for its own + SearchAsync call is reused here rather than this class creating its own, since it's + only needed for the one-time load. @@ -24559,6 +24633,20 @@ EmbeddingIndex's vectors (which are normalized the same way). + + + Detects when a free-text query names a specific poet and/or book/collection (e.g. + "در کدام شعر حافظ" -> حافظ, "در کدام بخش شاهنامه" -> شاهنامه) so search can be scoped to + just that poet/category instead of the whole corpus. + + Deliberately simple substring matching, not real NLP/NER — good enough for the common, + unambiguous case (a poet's distinctive nickname, or a book title effectively unique to one + poet, like شاهنامه), and safely conservative for the ambiguous case: a generic category + title shared by many poets (غزلیات appears for most of them) is only used as a scope if a + specific poet was ALSO named in the same query, narrowing which one is meant — otherwise + it's dropped rather than guessing which poet's غزلیات the person meant. + + url diff --git a/RMuseum/Services/Implementation/SemanticSearchService.cs b/RMuseum/Services/Implementation/SemanticSearchService.cs index 8faafc3b..1a523c46 100644 --- a/RMuseum/Services/Implementation/SemanticSearchService.cs +++ b/RMuseum/Services/Implementation/SemanticSearchService.cs @@ -103,8 +103,15 @@ namespace RMuseum.Services.Implementation HashSet allowedPoemIds = null; if (scopeCatId.HasValue) { + // A detected category like شاهنامه isn't where poems live directly — it's a + // book broken into many nested subcategories (individual kings/stories), with + // the actual poems several levels deeper. Matching only p.CatId == + // scopeCatId.Value (the original version of this code) found essentially + // nothing for exactly that reason — it needs every descendant category, not + // just the one that was named. + var descendantCatIds = await GetDescendantCategoryIdsAsync(context, scopeCatId.Value); var ids = await context.GanjoorPoems.AsNoTracking() - .Where(p => p.CatId == scopeCatId.Value) + .Where(p => descendantCatIds.Contains(p.CatId)) .Select(p => p.Id) .ToListAsync(); allowedPoemIds = new HashSet(ids); @@ -164,5 +171,37 @@ namespace RMuseum.Services.Implementation return response; } + + /// + /// Walks the whole category subtree rooted at rootCatId (breadth-first, level by level) + /// and returns every category id in it, including rootCatId itself. Needed because a + /// detected "book" category (شاهنامه, غزلیات, ...) is rarely where poems live directly — + /// it's typically broken into many nested subcategories, with the actual poems several + /// levels deeper. One query per depth level, not per node — a large book with many + /// subcategories still only costs as many round-trips as the tree is deep, not how many + /// nodes it has. + /// + private static async Task> GetDescendantCategoryIdsAsync(RMuseumDbContext context, int rootCatId) + { + var allCatIds = new HashSet { rootCatId }; + var frontier = new List { rootCatId }; + + while (frontier.Count > 0) + { + var children = await context.GanjoorCategories.AsNoTracking() + .Where(c => c.ParentId.HasValue && frontier.Contains(c.ParentId.Value)) + .Select(c => c.Id) + .ToListAsync(); + + var newIds = children.Where(id => !allCatIds.Contains(id)).ToList(); + foreach (var id in newIds) + { + allCatIds.Add(id); + } + frontier = newIds; + } + + return allCatIds; + } } }