divan/RMuseum/Services/Implementation/SemanticSearchService.cs
Hamid Reza Mohammadi 64fdb6318c semantic search
2026-09-11 16:45:47 +03:30

432 lines
23 KiB
C#
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

using Microsoft.EntityFrameworkCore;
using RMuseum.DbContext;
using RMuseum.Models.Ganjoor;
using RMuseum.Models.Ganjoor.SemanticSearch;
using RMuseum.Utils.SemanticSearch;
using System;
using System.Collections.Generic;
using System.Linq;
using System.Threading.Tasks;
namespace RMuseum.Services.Implementation
{
public interface ISemanticSearchService
{
Task<SemanticSearchResponseDto> SearchAsync(SemanticSearchRequestDto request);
/// <summary>
/// Best-effort — see the implementation for why this should never throw in a way the
/// caller needs to handle.
/// </summary>
Task ReportClickAsync(SemanticSearchClickDto click);
}
/// <summary>
/// Deliberately NOT a GanjoorService partial, unlike everything else in this project.
/// EmbeddingIndex (~530MB in memory) and QueryEmbedder (a loaded ONNX model) both need to be
/// true singletons — constructed once at startup, never per-request — while GanjoorService
/// and RMuseumDbContext are scoped per-request throughout this codebase. Injecting a
/// singleton's dependencies into a per-request class (or vice versa) is a real DI lifetime
/// bug, not just an inconsistency, so this stays a separate service registered as a
/// singleton itself (see INTEGRATION.md for the exact registration).
///
/// Depends on LazySemanticSearchResources rather than EmbeddingIndex/QueryEmbedder directly —
/// deliberately, after a production incident where eager, throwing DI factories for those two
/// meant a load failure (wrong/missing file paths) prevented GanjoorController itself from
/// being constructed, taking down every endpoint under /api/ganjoor with a 503, not just
/// semantic search. This class must never let a resource-loading failure become an unhandled
/// exception that propagates past SearchAsync — see the catch below.
///
/// LazyQueryScopeIndex is a SEPARATE lazy singleton from LazySemanticSearchResources on
/// purpose — poet/category name detection ("در کدام شعر حافظ") is a genuinely independent
/// concern from the embedding/model loading, with its own independent failure mode; a bug in
/// one must not be able to disable the other.
/// </summary>
public class SemanticSearchService : ISemanticSearchService
{
private const int DefaultTopK = 10;
private const int MaxTopK = 50;
private const int PreviewVerseCount = 4; // ~2 couplets - enough for a result-card preview, not the whole poem
private const int MaxVersesToScanForRelevance = 200; // bounded prefix to search for a relevant couplet within
private const int CandidatePoolMultiplier = 3; // how much larger than k the pre-re-rank candidate pool is
private const string AiGeneratedSummaryPrefix = "هوش مصنوعی:";
private readonly LazySemanticSearchResources _resources;
private readonly LazyQueryScopeIndex _queryScopeIndex;
public SemanticSearchService(LazySemanticSearchResources resources, LazyQueryScopeIndex queryScopeIndex)
{
_resources = resources;
_queryScopeIndex = queryScopeIndex;
}
public async Task<SemanticSearchResponseDto> SearchAsync(SemanticSearchRequestDto request)
{
if (request == null || string.IsNullOrWhiteSpace(request.Query))
throw new ArgumentException("query must not be empty", nameof(request));
if (!_resources.TryGetResources(out var embeddingIndex, out var queryEmbedder, out var error))
{
throw new SemanticSearchUnavailableException(
"Semantic search is not available right now" + (error != null ? $": {error}" : "."));
}
int k = request.TopK.GetValueOrDefault(DefaultTopK);
if (k <= 0 || k > MaxTopK)
k = DefaultTopK;
var response = new SemanticSearchResponseDto { Query = request.Query };
// a fresh, short-lived DbContext per call - same pattern GanjoorService's background
// jobs already use throughout this codebase, since a scoped/request DbContext can't
// be injected into this singleton service
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>()))
{
// An explicit request.PoetId/CatId (from a future UI, e.g. a poet picker) always
// wins; auto-detection only fills in whatever wasn't already specified. A
// detection failure here is swallowed on top of LazyQueryScopeIndex's own
// try/catch, as a second safety net - this is a nice-to-have, never worth
// failing the whole search over.
//
// request.DisableScopeDetection skips this block entirely - the "search globally
// instead" escape hatch for when detection guessed wrong (e.g. "شمع و پروانه" as
// a theme, not the specific book of that name by a different poet). An explicit
// PoetId/CatId still applies even here, since that's a deliberate ask, not a guess.
int? scopePoetId = request.PoetId;
int? scopeCatId = request.CatId;
if (!request.DisableScopeDetection)
{
try
{
var scopeIndex = await _queryScopeIndex.TryGetIndexAsync(context);
if (scopeIndex != null)
{
var detected = scopeIndex.DetectScope(request.Query);
if (detected.HasAny)
{
scopePoetId = scopePoetId ?? detected.PoetId;
scopeCatId = scopeCatId ?? detected.CatId;
response.DetectedPoetName = detected.PoetName;
response.DetectedCategoryName = detected.CategoryName;
}
}
}
catch (Exception)
{
// scope detection is best-effort only - fall through with no scope rather
// than fail the search
}
}
HashSet<int> allowedPoemIds = null;
if (scopeCatId.HasValue)
{
// A detected category like شاهنامه isn't where poems live directly — it's a
// book broken into many nested subcategories (individual kings/stories), with
// the actual poems several levels deeper. Matching only p.CatId ==
// scopeCatId.Value (the original version of this code) found essentially
// nothing for exactly that reason — it needs every descendant category, not
// just the one that was named.
var descendantCatIds = await GetDescendantCategoryIdsAsync(context, scopeCatId.Value);
var ids = await context.GanjoorPoems.AsNoTracking()
.Where(p => descendantCatIds.Contains(p.CatId))
.Select(p => p.Id)
.ToListAsync();
allowedPoemIds = new HashSet<int>(ids);
}
else if (scopePoetId.HasValue)
{
var ids = await context.GanjoorPoems.AsNoTracking()
.Where(p => p.Cat.PoetId == scopePoetId.Value)
.Select(p => p.Id)
.ToListAsync();
allowedPoemIds = new HashSet<int>(ids);
}
float[] queryVector = queryEmbedder.EmbedQuery(request.Query);
// Fetch a larger candidate pool than the final k, so there's room for the
// AI-summary re-ranking below to actually change the outcome — re-ranking a pool
// that's already exactly k results wide can only reorder them, never let a
// human-reviewed poem that was originally ranked k+1 overtake one that made the
// cut only because its (unreviewed) summary happened to score marginally higher.
int candidatePoolSize = Math.Min(k * CandidatePoolMultiplier, MaxTopK * CandidatePoolMultiplier);
List<(int PoemId, float Score)> candidates = embeddingIndex.FindTopSimilar(queryVector, candidatePoolSize, allowedPoemIds);
var candidateIds = candidates.Select(c => c.PoemId).ToList();
var poemsById = await context.GanjoorPoems.AsNoTracking()
.Where(p => candidateIds.Contains(p.Id))
.ToDictionaryAsync(p => p.Id);
// Gentle re-rank: a poem whose summary is still AI-generated and un-reviewed
// (still carries ganjoor-data's own "هوش مصنوعی:" prefix - removing it is part of
// that project's human-review/edit workflow) gets a small score penalty before
// final sorting. The DISPLAYED score stays the true, unpenalized cosine
// similarity (Score below uses c.Score, not AdjustedScore) - the penalty is
// purely an internal ordering nudge, not something that should make the shown
// similarity number stop meaning what it says.
float aiPenalty = _resources.AiSummaryScorePenalty;
List<(int PoemId, float Score)> topMatches = candidates
.Where(c => poemsById.ContainsKey(c.PoemId))
.Select(c => new
{
c.PoemId,
c.Score,
AdjustedScore = IsAiGeneratedSummary(poemsById[c.PoemId].PoemSummary) ? c.Score * aiPenalty : c.Score,
})
.OrderByDescending(x => x.AdjustedScore)
.Take(k)
.Select(x => (x.PoemId, x.Score))
.ToList();
// computed once for the whole request, not per result - the query doesn't change
// between results
var queryKeywords = ExtractKeywords(request.Query);
foreach (var match in topMatches)
{
// a poem id present in the embedding index but missing from the live DB
// (e.g. deleted/unpublished since the embeddings were generated) is silently
// skipped rather than failing the whole search for everyone else's results
if (!poemsById.TryGetValue(match.PoemId, out var poem))
continue;
// Fetch a bounded prefix of the poem's verses (not just the first
// PreviewVerseCount) so SelectPreviewVerses has real material to search
// through for a couplet that actually relates to the query, rather than
// always showing the poem's opening lines regardless of relevance. Capped at
// MaxVersesToScanForRelevance rather than the whole poem - covers the vast
// majority of ghazals/qasides/robaiyat in full, and still a reasonable bound
// for longer forms without pulling an unbounded number of rows per result.
var candidateVerses = await context.GanjoorVerses.AsNoTracking()
.Where(v => v.PoemId == poem.Id)
.OrderBy(v => v.VOrder)
.Take(MaxVersesToScanForRelevance)
.ToListAsync();
var verses = SelectPreviewVerses(candidateVerses, queryKeywords, PreviewVerseCount);
response.Results.Add(new SemanticSearchResultDto
{
PoemId = poem.Id,
Title = poem.Title,
FullTitle = poem.FullTitle,
FullUrl = poem.FullUrl,
PoemSummary = poem.PoemSummary,
Score = match.Score,
Verses = verses.Select(v => new SemanticSearchVerseDto
{
VOrder = v.VOrder,
Position = v.VersePosition.ToString(),
Text = v.Text,
}).ToList(),
});
}
// Best-effort logging, using the same context already open for this request - a
// logging failure must never fail a real search, so any exception here is
// swallowed and response.LogId simply stays null (ReportClickAsync already
// no-ops safely if it's ever called with a null/0 LogId from a search that
// couldn't be logged).
try
{
var log = new SemanticSearchQueryLog
{
Query = request.Query,
DateTimeUtc = DateTime.UtcNow,
RequestedTopK = k,
ResultCount = response.Results.Count,
TopResultScore = response.Results.Count > 0 ? (float?)response.Results[0].Score : null,
ScopeDetected = !string.IsNullOrEmpty(response.DetectedPoetName) || !string.IsNullOrEmpty(response.DetectedCategoryName),
DetectedPoetName = response.DetectedPoetName,
DetectedCategoryName = response.DetectedCategoryName,
ScopeDetectionDisabled = request.DisableScopeDetection,
};
context.SemanticSearchQueryLogs.Add(log);
await context.SaveChangesAsync();
response.LogId = log.Id;
}
catch (Exception)
{
// logging is best-effort only - never worth failing a real search over
}
}
return response;
}
public async Task ReportClickAsync(SemanticSearchClickDto click)
{
if (click == null || click.LogId <= 0)
return;
try
{
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>()))
{
var log = await context.SemanticSearchQueryLogs.FindAsync(click.LogId);
if (log != null)
{
log.ClickedPoemId = click.PoemId;
log.ClickedResultRank = click.Rank;
log.ClickedAtUtc = DateTime.UtcNow;
await context.SaveChangesAsync();
}
}
}
catch (Exception)
{
// click reporting is best-effort only - the click/navigation has already
// happened regardless of whether this write succeeds, never worth surfacing an
// error to the (fire-and-forget) caller for this
}
}
/// <summary>
/// ganjoor-data's own editing workflow requires this exact prefix be removed once a
/// human has reviewed/edited a poem's summary — its continued presence is a direct,
/// already-existing signal for "not yet human-reviewed," not something this project
/// invented or has to infer.
/// </summary>
private static bool IsAiGeneratedSummary(string poemSummary)
{
return !string.IsNullOrEmpty(poemSummary) &&
poemSummary.StartsWith(AiGeneratedSummaryPrefix, StringComparison.Ordinal);
}
/// <summary>
/// Common Persian function words stripped out before keyword-matching a query against
/// verse text (see SelectPreviewVerses) — the kind of words that appear in nearly every
/// query regardless of topic ("شعری در مورد ... پیدا کن") and would otherwise match
/// almost any couplet in almost any poem, defeating the whole point of looking for a
/// RELEVANT couplet rather than an arbitrary one. Not remotely exhaustive Persian
/// stopword coverage — just the words that actually show up in how people phrase this
/// kind of query, extended as real queries reveal gaps.
/// </summary>
private static readonly HashSet<string> PersianStopWords = new HashSet<string>
{
"شعر", "شعری", "شعرهای", "غزل", "غزلی", "قصیده", "رباعی", "مثنوی",
"در", "مورد", "به", "از", "با", "برای", "را", "که", "یک", "این", "آن",
"پیدا", "کن", "کنید", "کنم", "می‌خواهم", "میخواهم", "می‌خوام", "میخوام",
"هست", "است", "بود", "تا", "یا", "و", "چه", "چگونه", "کدام", "چطور",
"درباره", "دربارهٔ", "دربارۀ", "راجع", "بر", "روی", "های", "ها", "می",
"خوب", "چیزی", "بگو", "بده", "نشان", "لطفا", "لطفاً",
};
/// <summary>
/// Splits the query on whitespace (deliberately NOT on ZWNJ — "بی‌وفایی" should survive
/// as one token, not fracture into "بی" + "وفایی", where "بی" alone is a common enough
/// prefix to false-positive-match all over the place), strips surrounding punctuation,
/// drops stopwords and anything too short to be a meaningful keyword on its own.
/// </summary>
private static List<string> ExtractKeywords(string query)
{
if (string.IsNullOrWhiteSpace(query))
return new List<string>();
char[] punctuation = { '؟', '?', '.', '،', ',', '!', ':', ';', '«', '»', '"', '\'' };
return query
.Split(new[] { ' ' }, StringSplitOptions.RemoveEmptyEntries)
.Select(w => w.Trim(punctuation))
.Where(w => w.Length >= 2 && !PersianStopWords.Contains(w))
.Distinct()
.ToList();
}
/// <summary>
/// Looks for the couplet (Right/Left verse pair) whose combined text contains the most
/// query keywords, and returns it (plus, if there's room within previewVerseCount, the
/// couplet immediately following it, for a little reading continuity rather than a
/// single isolated pair). Falls back to the poem's opening verses — the previous,
/// always-the-same-lines behavior — if no keyword appears anywhere in the scanned
/// verses, or if there were no real keywords to search for at all (a query that was
/// entirely stopwords, or empty after stripping them).
///
/// Matches against CoupletSummary (when present) as well as the raw verse text — the
/// summary is clean, modern-language prose, while the verses themselves are archaic and
/// metaphorical and often won't literally contain a query's keywords even when the
/// couplet is genuinely on-topic. CoupletSummary is only ever read from the FIRST verse
/// of the pair (the "Right"/anchor position) — ganjoor-data stores it once per couplet
/// there, never on the second ("Left") verse; a small number of "Left" rows do carry a
/// stray value (a data anomaly, not a second legitimate copy), and this deliberately
/// never reads it from that position, matching how the data is actually meant to be laid
/// out rather than how a few rows happen to look.
/// </summary>
private static List<GanjoorVerse> SelectPreviewVerses(List<GanjoorVerse> allVerses, List<string> keywords, int previewVerseCount)
{
if (keywords.Count > 0)
{
int bestScore = 0;
int bestIndex = -1;
for (int i = 0; i < allVerses.Count - 1; i++)
{
if (allVerses[i].VersePosition.ToString() != "Right" || allVerses[i + 1].VersePosition.ToString() != "Left")
continue;
string coupletText = allVerses[i].Text + " " + allVerses[i + 1].Text;
if (!string.IsNullOrEmpty(allVerses[i].CoupletSummary))
{
coupletText += " " + allVerses[i].CoupletSummary;
}
int score = keywords.Count(k => coupletText.Contains(k));
if (score > bestScore)
{
bestScore = score;
bestIndex = i;
}
}
if (bestIndex >= 0)
{
var result = new List<GanjoorVerse> { allVerses[bestIndex], allVerses[bestIndex + 1] };
int next = bestIndex + 2;
while (result.Count < previewVerseCount && next < allVerses.Count)
{
result.Add(allVerses[next]);
next++;
}
return result;
}
}
// fallback: the poem's opening verses, same as the original behavior
return allVerses.Take(previewVerseCount).ToList();
}
/// <summary>
/// Walks the whole category subtree rooted at rootCatId (breadth-first, level by level)
/// and returns every category id in it, including rootCatId itself. Needed because a
/// detected "book" category (شاهنامه, غزلیات, ...) is rarely where poems live directly —
/// it's typically broken into many nested subcategories, with the actual poems several
/// levels deeper. One query per depth level, not per node — a large book with many
/// subcategories still only costs as many round-trips as the tree is deep, not how many
/// nodes it has.
/// </summary>
private static async Task<HashSet<int>> GetDescendantCategoryIdsAsync(RMuseumDbContext context, int rootCatId)
{
var allCatIds = new HashSet<int> { rootCatId };
var frontier = new List<int> { rootCatId };
while (frontier.Count > 0)
{
var children = await context.GanjoorCategories.AsNoTracking()
.Where(c => c.ParentId.HasValue && frontier.Contains(c.ParentId.Value))
.Select(c => c.Id)
.ToListAsync();
var newIds = children.Where(id => !allCatIds.Contains(id)).ToList();
foreach (var id in newIds)
{
allCatIds.Add(id);
}
frontier = newIds;
}
return allCatIds;
}
}
}