semantic search
This commit is contained in:
parent
e63c8c5e9d
commit
851ea607a2
@ -67,9 +67,16 @@
|
|||||||
},
|
},
|
||||||
error: function (xhr) {
|
error: function (xhr) {
|
||||||
status.style.display = "block";
|
status.style.display = "block";
|
||||||
status.textContent = xhr.status === 503
|
if (xhr.status === 503) {
|
||||||
? "در حال حاضر جستوجوی معنایی در دسترس نیست. لطفاً کمی بعد دوباره امتحان کنید."
|
status.textContent = "در حال حاضر جستوجوی معنایی در دسترس نیست. لطفاً کمی بعد دوباره امتحان کنید.";
|
||||||
: "خطایی رخ داد. لطفاً دوباره تلاش کنید.";
|
} else if (xhr.responseText) {
|
||||||
|
// show the real server error text rather than a generic message -
|
||||||
|
// whatever RMuseum actually said, so a real failure is diagnosable
|
||||||
|
// from what's on screen instead of needing another round-trip
|
||||||
|
status.textContent = "خطا: " + xhr.responseText;
|
||||||
|
} else {
|
||||||
|
status.textContent = "خطایی رخ داد (کد " + xhr.status + "). لطفاً دوباره تلاش کنید.";
|
||||||
|
}
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|||||||
@ -11377,12 +11377,43 @@
|
|||||||
how many results to return; a sane default is applied server-side if omitted/invalid
|
how many results to return; a sane default is applied server-side if omitted/invalid
|
||||||
</summary>
|
</summary>
|
||||||
</member>
|
</member>
|
||||||
|
<member name="P:RMuseum.Models.Ganjoor.SemanticSearch.SemanticSearchRequestDto.PoetId">
|
||||||
|
<summary>
|
||||||
|
Optional explicit scope, for a future UI (e.g. a poet picker) that wants to restrict
|
||||||
|
search without relying on auto-detection from the query text. Auto-detection (see
|
||||||
|
SemanticSearchService.DetectQueryScope) still runs even when these are set — an
|
||||||
|
explicit PoetId/CatId narrows further, it doesn't replace detection.
|
||||||
|
</summary>
|
||||||
|
</member>
|
||||||
|
<member name="P:RMuseum.Models.Ganjoor.SemanticSearch.SemanticSearchVerseDto.Position">
|
||||||
|
<summary>
|
||||||
|
matches GanjoorVerse.VersePosition.ToString() (e.g. "Right"/"Left") - same convention
|
||||||
|
already used by the main ganjoor-data export, so any existing hemistich-pairing
|
||||||
|
frontend code (see mini-ganjoor's renderVerses) can be reused as-is.
|
||||||
|
</summary>
|
||||||
|
</member>
|
||||||
|
<member name="P:RMuseum.Models.Ganjoor.SemanticSearch.SemanticSearchResultDto.Verses">
|
||||||
|
<summary>
|
||||||
|
A short preview from the poem's actual text - the first couple of couplets, in
|
||||||
|
original verse order. Not the whole poem; just enough for a result card.
|
||||||
|
</summary>
|
||||||
|
</member>
|
||||||
<member name="P:RMuseum.Models.Ganjoor.SemanticSearch.SemanticSearchResultDto.Score">
|
<member name="P:RMuseum.Models.Ganjoor.SemanticSearch.SemanticSearchResultDto.Score">
|
||||||
<summary>
|
<summary>
|
||||||
cosine similarity, 0..1 for these normalized vectors (in practice results cluster in
|
cosine similarity, 0..1 for these normalized vectors (in practice results cluster in
|
||||||
a narrower band - this is a relative ranking signal, not a calibrated probability)
|
a narrower band - this is a relative ranking signal, not a calibrated probability)
|
||||||
</summary>
|
</summary>
|
||||||
</member>
|
</member>
|
||||||
|
<member name="P:RMuseum.Models.Ganjoor.SemanticSearch.SemanticSearchResponseDto.DetectedPoetName">
|
||||||
|
<summary>
|
||||||
|
If the query text was recognized as referring to a specific poet and/or
|
||||||
|
category/book (e.g. "در کدام شعر حافظ" -> حافظ, "در کدام بخش شاهنامه" -> شاهنامه),
|
||||||
|
search was restricted to that scope and these are populated so the UI can show the
|
||||||
|
user what was detected ("نتایج محدود به: حافظ") rather than silently filtering.
|
||||||
|
Null/null if nothing was detected (or an explicit PoetId/CatId narrowed things
|
||||||
|
without any name being recognized in the free text).
|
||||||
|
</summary>
|
||||||
|
</member>
|
||||||
<member name="T:RMuseum.Models.Ganjoor.UpdatingRelSectsLog">
|
<member name="T:RMuseum.Models.Ganjoor.UpdatingRelSectsLog">
|
||||||
<summary>
|
<summary>
|
||||||
Updating related sections logs
|
Updating related sections logs
|
||||||
@ -23104,8 +23135,24 @@
|
|||||||
being constructed, taking down every endpoint under /api/ganjoor with a 503, not just
|
being constructed, taking down every endpoint under /api/ganjoor with a 503, not just
|
||||||
semantic search. This class must never let a resource-loading failure become an unhandled
|
semantic search. This class must never let a resource-loading failure become an unhandled
|
||||||
exception that propagates past SearchAsync — see the catch below.
|
exception that propagates past SearchAsync — see the catch below.
|
||||||
|
|
||||||
|
LazyQueryScopeIndex is a SEPARATE lazy singleton from LazySemanticSearchResources on
|
||||||
|
purpose — poet/category name detection ("در کدام شعر حافظ") is a genuinely independent
|
||||||
|
concern from the embedding/model loading, with its own independent failure mode; a bug in
|
||||||
|
one must not be able to disable the other.
|
||||||
</summary>
|
</summary>
|
||||||
</member>
|
</member>
|
||||||
|
<member name="M:RMuseum.Services.Implementation.SemanticSearchService.GetDescendantCategoryIdsAsync(RMuseum.DbContext.RMuseumDbContext,System.Int32)">
|
||||||
|
<summary>
|
||||||
|
Walks the whole category subtree rooted at rootCatId (breadth-first, level by level)
|
||||||
|
and returns every category id in it, including rootCatId itself. Needed because a
|
||||||
|
detected "book" category (شاهنامه, غزلیات, ...) is rarely where poems live directly —
|
||||||
|
it's typically broken into many nested subcategories, with the actual poems several
|
||||||
|
levels deeper. One query per depth level, not per node — a large book with many
|
||||||
|
subcategories still only costs as many round-trips as the tree is deep, not how many
|
||||||
|
nodes it has.
|
||||||
|
</summary>
|
||||||
|
</member>
|
||||||
<member name="T:RMuseum.Services.Implementation.SiteBannersService">
|
<member name="T:RMuseum.Services.Implementation.SiteBannersService">
|
||||||
<summary>
|
<summary>
|
||||||
ganjoor.net banners service
|
ganjoor.net banners service
|
||||||
@ -24462,12 +24509,39 @@
|
|||||||
memory, not something to redo per-request.
|
memory, not something to redo per-request.
|
||||||
</summary>
|
</summary>
|
||||||
</member>
|
</member>
|
||||||
<member name="M:RMuseum.Utils.SemanticSearch.EmbeddingIndex.FindTopSimilar(System.ReadOnlySpan{System.Single},System.Int32)">
|
<member name="M:RMuseum.Utils.SemanticSearch.EmbeddingIndex.FindTopSimilar(System.ReadOnlySpan{System.Single},System.Int32,System.Collections.Generic.ISet{System.Int32})">
|
||||||
|
<summary>
|
||||||
|
Returns the topK poem ids most similar to queryVector, ranked descending by cosine
|
||||||
|
similarity. queryVector must already be the SAME dimension as this index and, for the
|
||||||
|
score to mean what it claims (a true cosine similarity), should already be
|
||||||
|
L2-normalized the same way the indexed vectors are — see QueryEmbedder.
|
||||||
|
|
||||||
|
If allowedPoemIds is provided (non-null), only poems in that set are eligible —
|
||||||
|
everything else is scored as excluded and can never appear in the results, however
|
||||||
|
similar it might be. Used for scoped search ("در کدام شعر حافظ" -> restrict to
|
||||||
|
Hafez's poems) — still a full scan either way, just with cheap early-outs for
|
||||||
|
excluded rows, since this corpus is small enough that a full scan is fast regardless.
|
||||||
|
</summary>
|
||||||
|
</member>
|
||||||
|
<member name="T:RMuseum.Utils.SemanticSearch.LazyQueryScopeIndex">
|
||||||
|
<summary>
|
||||||
|
Same "load once, never throw" pattern as LazySemanticSearchResources, but deliberately a
|
||||||
|
SEPARATE singleton with its own independent failure domain: if the poet/category name
|
||||||
|
lookup fails to load for any reason, that must only disable scope auto-detection
|
||||||
|
("در کدام شعر حافظ" -> unscoped, searches everything) — it must never affect the embedding
|
||||||
|
index/model loading or plain (unscoped) search, which are a completely different concern.
|
||||||
|
|
||||||
|
Uses a SemaphoreSlim rather than a plain `lock`, since the actual load is async (DB
|
||||||
|
queries) and you can't `await` inside a `lock` block.
|
||||||
|
</summary>
|
||||||
|
</member>
|
||||||
|
<member name="M:RMuseum.Utils.SemanticSearch.LazyQueryScopeIndex.TryGetIndexAsync(RMuseum.DbContext.RMuseumDbContext)">
|
||||||
<summary>
|
<summary>
|
||||||
Returns the topK poem ids most similar to queryVector, ranked descending by cosine
|
Loads (once) and returns the scope index, or null if loading failed or hasn't
|
||||||
similarity. queryVector must already be the SAME dimension as this index and, for the
|
succeeded yet — NEVER throws. The passed-in context is only actually used on the
|
||||||
score to mean what it claims (a true cosine similarity), should already be
|
first call that does real work; a context created by the caller for its own
|
||||||
L2-normalized the same way the indexed vectors are — see QueryEmbedder.
|
SearchAsync call is reused here rather than this class creating its own, since it's
|
||||||
|
only needed for the one-time load.
|
||||||
</summary>
|
</summary>
|
||||||
</member>
|
</member>
|
||||||
<member name="T:RMuseum.Utils.SemanticSearch.LazySemanticSearchResources">
|
<member name="T:RMuseum.Utils.SemanticSearch.LazySemanticSearchResources">
|
||||||
@ -24559,6 +24633,20 @@
|
|||||||
EmbeddingIndex's vectors (which are normalized the same way).
|
EmbeddingIndex's vectors (which are normalized the same way).
|
||||||
</summary>
|
</summary>
|
||||||
</member>
|
</member>
|
||||||
|
<member name="T:RMuseum.Utils.SemanticSearch.QueryScopeIndex">
|
||||||
|
<summary>
|
||||||
|
Detects when a free-text query names a specific poet and/or book/collection (e.g.
|
||||||
|
"در کدام شعر حافظ" -> حافظ, "در کدام بخش شاهنامه" -> شاهنامه) so search can be scoped to
|
||||||
|
just that poet/category instead of the whole corpus.
|
||||||
|
|
||||||
|
Deliberately simple substring matching, not real NLP/NER — good enough for the common,
|
||||||
|
unambiguous case (a poet's distinctive nickname, or a book title effectively unique to one
|
||||||
|
poet, like شاهنامه), and safely conservative for the ambiguous case: a generic category
|
||||||
|
title shared by many poets (غزلیات appears for most of them) is only used as a scope if a
|
||||||
|
specific poet was ALSO named in the same query, narrowing which one is meant — otherwise
|
||||||
|
it's dropped rather than guessing which poet's غزلیات the person meant.
|
||||||
|
</summary>
|
||||||
|
</member>
|
||||||
<member name="P:RMuseum.WebServiceUrl.Url">
|
<member name="P:RMuseum.WebServiceUrl.Url">
|
||||||
<summary>
|
<summary>
|
||||||
url
|
url
|
||||||
|
|||||||
@ -103,8 +103,15 @@ namespace RMuseum.Services.Implementation
|
|||||||
HashSet<int> allowedPoemIds = null;
|
HashSet<int> allowedPoemIds = null;
|
||||||
if (scopeCatId.HasValue)
|
if (scopeCatId.HasValue)
|
||||||
{
|
{
|
||||||
|
// A detected category like شاهنامه isn't where poems live directly — it's a
|
||||||
|
// book broken into many nested subcategories (individual kings/stories), with
|
||||||
|
// the actual poems several levels deeper. Matching only p.CatId ==
|
||||||
|
// scopeCatId.Value (the original version of this code) found essentially
|
||||||
|
// nothing for exactly that reason — it needs every descendant category, not
|
||||||
|
// just the one that was named.
|
||||||
|
var descendantCatIds = await GetDescendantCategoryIdsAsync(context, scopeCatId.Value);
|
||||||
var ids = await context.GanjoorPoems.AsNoTracking()
|
var ids = await context.GanjoorPoems.AsNoTracking()
|
||||||
.Where(p => p.CatId == scopeCatId.Value)
|
.Where(p => descendantCatIds.Contains(p.CatId))
|
||||||
.Select(p => p.Id)
|
.Select(p => p.Id)
|
||||||
.ToListAsync();
|
.ToListAsync();
|
||||||
allowedPoemIds = new HashSet<int>(ids);
|
allowedPoemIds = new HashSet<int>(ids);
|
||||||
@ -164,5 +171,37 @@ namespace RMuseum.Services.Implementation
|
|||||||
|
|
||||||
return response;
|
return response;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// Walks the whole category subtree rooted at rootCatId (breadth-first, level by level)
|
||||||
|
/// and returns every category id in it, including rootCatId itself. Needed because a
|
||||||
|
/// detected "book" category (شاهنامه, غزلیات, ...) is rarely where poems live directly —
|
||||||
|
/// it's typically broken into many nested subcategories, with the actual poems several
|
||||||
|
/// levels deeper. One query per depth level, not per node — a large book with many
|
||||||
|
/// subcategories still only costs as many round-trips as the tree is deep, not how many
|
||||||
|
/// nodes it has.
|
||||||
|
/// </summary>
|
||||||
|
private static async Task<HashSet<int>> GetDescendantCategoryIdsAsync(RMuseumDbContext context, int rootCatId)
|
||||||
|
{
|
||||||
|
var allCatIds = new HashSet<int> { rootCatId };
|
||||||
|
var frontier = new List<int> { rootCatId };
|
||||||
|
|
||||||
|
while (frontier.Count > 0)
|
||||||
|
{
|
||||||
|
var children = await context.GanjoorCategories.AsNoTracking()
|
||||||
|
.Where(c => c.ParentId.HasValue && frontier.Contains(c.ParentId.Value))
|
||||||
|
.Select(c => c.Id)
|
||||||
|
.ToListAsync();
|
||||||
|
|
||||||
|
var newIds = children.Where(id => !allCatIds.Contains(id)).ToList();
|
||||||
|
foreach (var id in newIds)
|
||||||
|
{
|
||||||
|
allCatIds.Add(id);
|
||||||
|
}
|
||||||
|
frontier = newIds;
|
||||||
|
}
|
||||||
|
|
||||||
|
return allCatIds;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user