semantic search
This commit is contained in:
parent
e63c8c5e9d
commit
851ea607a2
@ -67,9 +67,16 @@
|
||||
},
|
||||
error: function (xhr) {
|
||||
status.style.display = "block";
|
||||
status.textContent = xhr.status === 503
|
||||
? "در حال حاضر جستوجوی معنایی در دسترس نیست. لطفاً کمی بعد دوباره امتحان کنید."
|
||||
: "خطایی رخ داد. لطفاً دوباره تلاش کنید.";
|
||||
if (xhr.status === 503) {
|
||||
status.textContent = "در حال حاضر جستوجوی معنایی در دسترس نیست. لطفاً کمی بعد دوباره امتحان کنید.";
|
||||
} else if (xhr.responseText) {
|
||||
// show the real server error text rather than a generic message -
|
||||
// whatever RMuseum actually said, so a real failure is diagnosable
|
||||
// from what's on screen instead of needing another round-trip
|
||||
status.textContent = "خطا: " + xhr.responseText;
|
||||
} else {
|
||||
status.textContent = "خطایی رخ داد (کد " + xhr.status + "). لطفاً دوباره تلاش کنید.";
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
@ -11377,12 +11377,43 @@
|
||||
how many results to return; a sane default is applied server-side if omitted/invalid
|
||||
</summary>
|
||||
</member>
|
||||
<member name="P:RMuseum.Models.Ganjoor.SemanticSearch.SemanticSearchRequestDto.PoetId">
|
||||
<summary>
|
||||
Optional explicit scope, for a future UI (e.g. a poet picker) that wants to restrict
|
||||
search without relying on auto-detection from the query text. Auto-detection (see
|
||||
SemanticSearchService.DetectQueryScope) still runs even when these are set — an
|
||||
explicit PoetId/CatId narrows further, it doesn't replace detection.
|
||||
</summary>
|
||||
</member>
|
||||
<member name="P:RMuseum.Models.Ganjoor.SemanticSearch.SemanticSearchVerseDto.Position">
|
||||
<summary>
|
||||
matches GanjoorVerse.VersePosition.ToString() (e.g. "Right"/"Left") - same convention
|
||||
already used by the main ganjoor-data export, so any existing hemistich-pairing
|
||||
frontend code (see mini-ganjoor's renderVerses) can be reused as-is.
|
||||
</summary>
|
||||
</member>
|
||||
<member name="P:RMuseum.Models.Ganjoor.SemanticSearch.SemanticSearchResultDto.Verses">
|
||||
<summary>
|
||||
A short preview from the poem's actual text - the first couple of couplets, in
|
||||
original verse order. Not the whole poem; just enough for a result card.
|
||||
</summary>
|
||||
</member>
|
||||
<member name="P:RMuseum.Models.Ganjoor.SemanticSearch.SemanticSearchResultDto.Score">
|
||||
<summary>
|
||||
cosine similarity, 0..1 for these normalized vectors (in practice results cluster in
|
||||
a narrower band - this is a relative ranking signal, not a calibrated probability)
|
||||
</summary>
|
||||
</member>
|
||||
<member name="P:RMuseum.Models.Ganjoor.SemanticSearch.SemanticSearchResponseDto.DetectedPoetName">
|
||||
<summary>
|
||||
If the query text was recognized as referring to a specific poet and/or
|
||||
category/book (e.g. "در کدام شعر حافظ" -> حافظ, "در کدام بخش شاهنامه" -> شاهنامه),
|
||||
search was restricted to that scope and these are populated so the UI can show the
|
||||
user what was detected ("نتایج محدود به: حافظ") rather than silently filtering.
|
||||
Null/null if nothing was detected (or an explicit PoetId/CatId narrowed things
|
||||
without any name being recognized in the free text).
|
||||
</summary>
|
||||
</member>
|
||||
<member name="T:RMuseum.Models.Ganjoor.UpdatingRelSectsLog">
|
||||
<summary>
|
||||
Updating related sections logs
|
||||
@ -23104,8 +23135,24 @@
|
||||
being constructed, taking down every endpoint under /api/ganjoor with a 503, not just
|
||||
semantic search. This class must never let a resource-loading failure become an unhandled
|
||||
exception that propagates past SearchAsync — see the catch below.
|
||||
|
||||
LazyQueryScopeIndex is a SEPARATE lazy singleton from LazySemanticSearchResources on
|
||||
purpose — poet/category name detection ("در کدام شعر حافظ") is a genuinely independent
|
||||
concern from the embedding/model loading, with its own independent failure mode; a bug in
|
||||
one must not be able to disable the other.
|
||||
</summary>
|
||||
</member>
|
||||
<member name="M:RMuseum.Services.Implementation.SemanticSearchService.GetDescendantCategoryIdsAsync(RMuseum.DbContext.RMuseumDbContext,System.Int32)">
|
||||
<summary>
|
||||
Walks the whole category subtree rooted at rootCatId (breadth-first, level by level)
|
||||
and returns every category id in it, including rootCatId itself. Needed because a
|
||||
detected "book" category (شاهنامه, غزلیات, ...) is rarely where poems live directly —
|
||||
it's typically broken into many nested subcategories, with the actual poems several
|
||||
levels deeper. One query per depth level, not per node — a large book with many
|
||||
subcategories still only costs as many round-trips as the tree is deep, not how many
|
||||
nodes it has.
|
||||
</summary>
|
||||
</member>
|
||||
<member name="T:RMuseum.Services.Implementation.SiteBannersService">
|
||||
<summary>
|
||||
ganjoor.net banners service
|
||||
@ -24462,12 +24509,39 @@
|
||||
memory, not something to redo per-request.
|
||||
</summary>
|
||||
</member>
|
||||
<member name="M:RMuseum.Utils.SemanticSearch.EmbeddingIndex.FindTopSimilar(System.ReadOnlySpan{System.Single},System.Int32)">
|
||||
<member name="M:RMuseum.Utils.SemanticSearch.EmbeddingIndex.FindTopSimilar(System.ReadOnlySpan{System.Single},System.Int32,System.Collections.Generic.ISet{System.Int32})">
|
||||
<summary>
|
||||
Returns the topK poem ids most similar to queryVector, ranked descending by cosine
|
||||
similarity. queryVector must already be the SAME dimension as this index and, for the
|
||||
score to mean what it claims (a true cosine similarity), should already be
|
||||
L2-normalized the same way the indexed vectors are — see QueryEmbedder.
|
||||
|
||||
If allowedPoemIds is provided (non-null), only poems in that set are eligible —
|
||||
everything else is scored as excluded and can never appear in the results, however
|
||||
similar it might be. Used for scoped search ("در کدام شعر حافظ" -> restrict to
|
||||
Hafez's poems) — still a full scan either way, just with cheap early-outs for
|
||||
excluded rows, since this corpus is small enough that a full scan is fast regardless.
|
||||
</summary>
|
||||
</member>
|
||||
<member name="T:RMuseum.Utils.SemanticSearch.LazyQueryScopeIndex">
|
||||
<summary>
|
||||
Same "load once, never throw" pattern as LazySemanticSearchResources, but deliberately a
|
||||
SEPARATE singleton with its own independent failure domain: if the poet/category name
|
||||
lookup fails to load for any reason, that must only disable scope auto-detection
|
||||
("در کدام شعر حافظ" -> unscoped, searches everything) — it must never affect the embedding
|
||||
index/model loading or plain (unscoped) search, which are a completely different concern.
|
||||
|
||||
Uses a SemaphoreSlim rather than a plain `lock`, since the actual load is async (DB
|
||||
queries) and you can't `await` inside a `lock` block.
|
||||
</summary>
|
||||
</member>
|
||||
<member name="M:RMuseum.Utils.SemanticSearch.LazyQueryScopeIndex.TryGetIndexAsync(RMuseum.DbContext.RMuseumDbContext)">
|
||||
<summary>
|
||||
Returns the topK poem ids most similar to queryVector, ranked descending by cosine
|
||||
similarity. queryVector must already be the SAME dimension as this index and, for the
|
||||
score to mean what it claims (a true cosine similarity), should already be
|
||||
L2-normalized the same way the indexed vectors are — see QueryEmbedder.
|
||||
Loads (once) and returns the scope index, or null if loading failed or hasn't
|
||||
succeeded yet — NEVER throws. The passed-in context is only actually used on the
|
||||
first call that does real work; a context created by the caller for its own
|
||||
SearchAsync call is reused here rather than this class creating its own, since it's
|
||||
only needed for the one-time load.
|
||||
</summary>
|
||||
</member>
|
||||
<member name="T:RMuseum.Utils.SemanticSearch.LazySemanticSearchResources">
|
||||
@ -24559,6 +24633,20 @@
|
||||
EmbeddingIndex's vectors (which are normalized the same way).
|
||||
</summary>
|
||||
</member>
|
||||
<member name="T:RMuseum.Utils.SemanticSearch.QueryScopeIndex">
|
||||
<summary>
|
||||
Detects when a free-text query names a specific poet and/or book/collection (e.g.
|
||||
"در کدام شعر حافظ" -> حافظ, "در کدام بخش شاهنامه" -> شاهنامه) so search can be scoped to
|
||||
just that poet/category instead of the whole corpus.
|
||||
|
||||
Deliberately simple substring matching, not real NLP/NER — good enough for the common,
|
||||
unambiguous case (a poet's distinctive nickname, or a book title effectively unique to one
|
||||
poet, like شاهنامه), and safely conservative for the ambiguous case: a generic category
|
||||
title shared by many poets (غزلیات appears for most of them) is only used as a scope if a
|
||||
specific poet was ALSO named in the same query, narrowing which one is meant — otherwise
|
||||
it's dropped rather than guessing which poet's غزلیات the person meant.
|
||||
</summary>
|
||||
</member>
|
||||
<member name="P:RMuseum.WebServiceUrl.Url">
|
||||
<summary>
|
||||
url
|
||||
|
||||
@ -103,8 +103,15 @@ namespace RMuseum.Services.Implementation
|
||||
HashSet<int> allowedPoemIds = null;
|
||||
if (scopeCatId.HasValue)
|
||||
{
|
||||
// A detected category like شاهنامه isn't where poems live directly — it's a
|
||||
// book broken into many nested subcategories (individual kings/stories), with
|
||||
// the actual poems several levels deeper. Matching only p.CatId ==
|
||||
// scopeCatId.Value (the original version of this code) found essentially
|
||||
// nothing for exactly that reason — it needs every descendant category, not
|
||||
// just the one that was named.
|
||||
var descendantCatIds = await GetDescendantCategoryIdsAsync(context, scopeCatId.Value);
|
||||
var ids = await context.GanjoorPoems.AsNoTracking()
|
||||
.Where(p => p.CatId == scopeCatId.Value)
|
||||
.Where(p => descendantCatIds.Contains(p.CatId))
|
||||
.Select(p => p.Id)
|
||||
.ToListAsync();
|
||||
allowedPoemIds = new HashSet<int>(ids);
|
||||
@ -164,5 +171,37 @@ namespace RMuseum.Services.Implementation
|
||||
|
||||
return response;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Walks the whole category subtree rooted at rootCatId (breadth-first, level by level)
|
||||
/// and returns every category id in it, including rootCatId itself. Needed because a
|
||||
/// detected "book" category (شاهنامه, غزلیات, ...) is rarely where poems live directly —
|
||||
/// it's typically broken into many nested subcategories, with the actual poems several
|
||||
/// levels deeper. One query per depth level, not per node — a large book with many
|
||||
/// subcategories still only costs as many round-trips as the tree is deep, not how many
|
||||
/// nodes it has.
|
||||
/// </summary>
|
||||
private static async Task<HashSet<int>> GetDescendantCategoryIdsAsync(RMuseumDbContext context, int rootCatId)
|
||||
{
|
||||
var allCatIds = new HashSet<int> { rootCatId };
|
||||
var frontier = new List<int> { rootCatId };
|
||||
|
||||
while (frontier.Count > 0)
|
||||
{
|
||||
var children = await context.GanjoorCategories.AsNoTracking()
|
||||
.Where(c => c.ParentId.HasValue && frontier.Contains(c.ParentId.Value))
|
||||
.Select(c => c.Id)
|
||||
.ToListAsync();
|
||||
|
||||
var newIds = children.Where(id => !allCatIds.Contains(id)).ToList();
|
||||
foreach (var id in newIds)
|
||||
{
|
||||
allCatIds.Add(id);
|
||||
}
|
||||
frontier = newIds;
|
||||
}
|
||||
|
||||
return allCatIds;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Loading…
Reference in New Issue
Block a user