using Microsoft.EntityFrameworkCore; using RMuseum.DbContext; using RMuseum.Models.Diwan; using RMuseum.Models.Diwan.SemanticSearch; using RMuseum.Utils.SemanticSearch; using System; using System.Collections.Generic; using System.Linq; using System.Threading.Tasks; namespace RMuseum.Services.Implementation { public interface ISemanticSearchService { Task SearchAsync(SemanticSearchRequestDto request); /// /// Best-effort — see the implementation for why this should never throw in a way the /// caller needs to handle. /// Task ReportClickAsync(SemanticSearchClickDto click); } /// /// Deliberately NOT a DiwanService partial, unlike everything else in this project. /// EmbeddingIndex (~530MB in memory) and QueryEmbedder (a loaded ONNX model) both need to be /// true singletons — constructed once at startup, never per-request — while DiwanService /// and RMuseumDbContext are scoped per-request throughout this codebase. Injecting a /// singleton's dependencies into a per-request class (or vice versa) is a real DI lifetime /// bug, not just an inconsistency, so this stays a separate service registered as a /// singleton itself (see INTEGRATION.md for the exact registration). /// /// Depends on LazySemanticSearchResources rather than EmbeddingIndex/QueryEmbedder directly — /// deliberately, after a production incident where eager, throwing DI factories for those two /// meant a load failure (wrong/missing file paths) prevented DiwanController itself from /// being constructed, taking down every endpoint under /api/diwan with a 503, not just /// semantic search. This class must never let a resource-loading failure become an unhandled /// exception that propagates past SearchAsync — see the catch below. /// /// LazyQueryScopeIndex is a SEPARATE lazy singleton from LazySemanticSearchResources on /// purpose — poet/category name detection ("در کدام شعر حافظ") is a genuinely independent /// concern from the embedding/model loading, with its own independent failure mode; a bug in /// one must not be able to disable the other. /// public class SemanticSearchService : ISemanticSearchService { private const int DefaultTopK = 10; private const int MaxTopK = 50; private const int PreviewVerseCount = 4; // ~2 couplets - enough for a result-card preview, not the whole poem private const int MaxVersesToScanForRelevance = 200; // bounded prefix to search for a relevant couplet within private const int CandidatePoolMultiplier = 3; // how much larger than k the pre-re-rank candidate pool is private const string AiGeneratedSummaryPrefix = "هوش مصنوعی:"; private readonly LazySemanticSearchResources _resources; private readonly LazyQueryScopeIndex _queryScopeIndex; public SemanticSearchService(LazySemanticSearchResources resources, LazyQueryScopeIndex queryScopeIndex) { _resources = resources; _queryScopeIndex = queryScopeIndex; } public async Task SearchAsync(SemanticSearchRequestDto request) { if (request == null || string.IsNullOrWhiteSpace(request.Query)) throw new ArgumentException("query must not be empty", nameof(request)); if (!_resources.TryGetResources(out var embeddingIndex, out var queryEmbedder, out var error)) { throw new SemanticSearchUnavailableException( "Semantic search is not available right now" + (error != null ? $": {error}" : ".")); } int k = request.TopK.GetValueOrDefault(DefaultTopK); if (k <= 0 || k > MaxTopK) k = DefaultTopK; var response = new SemanticSearchResponseDto { Query = request.Query }; // a fresh, short-lived DbContext per call - same pattern DiwanService's background // jobs already use throughout this codebase, since a scoped/request DbContext can't // be injected into this singleton service using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions())) { // An explicit request.PoetId/CatId (from a future UI, e.g. a poet picker) always // wins; auto-detection only fills in whatever wasn't already specified. A // detection failure here is swallowed on top of LazyQueryScopeIndex's own // try/catch, as a second safety net - this is a nice-to-have, never worth // failing the whole search over. // // request.DisableScopeDetection skips this block entirely - the "search globally // instead" escape hatch for when detection guessed wrong (e.g. "شمع و پروانه" as // a theme, not the specific book of that name by a different poet). An explicit // PoetId/CatId still applies even here, since that's a deliberate ask, not a guess. int? scopePoetId = request.PoetId; int? scopeCatId = request.CatId; if (!request.DisableScopeDetection) { try { var scopeIndex = await _queryScopeIndex.TryGetIndexAsync(context); if (scopeIndex != null) { var detected = scopeIndex.DetectScope(request.Query); if (detected.HasAny) { scopePoetId = scopePoetId ?? detected.PoetId; scopeCatId = scopeCatId ?? detected.CatId; response.DetectedPoetName = detected.PoetName; response.DetectedCategoryName = detected.CategoryName; } } } catch (Exception) { // scope detection is best-effort only - fall through with no scope rather // than fail the search } } HashSet allowedPoemIds = null; if (scopeCatId.HasValue) { // A detected category like شاهنامه isn't where poems live directly — it's a // book broken into many nested subcategories (individual kings/stories), with // the actual poems several levels deeper. Matching only p.CatId == // scopeCatId.Value (the original version of this code) found essentially // nothing for exactly that reason — it needs every descendant category, not // just the one that was named. var descendantCatIds = await GetDescendantCategoryIdsAsync(context, scopeCatId.Value); var ids = await context.DiwanPoems.AsNoTracking() .Where(p => descendantCatIds.Contains(p.CatId)) .Select(p => p.Id) .ToListAsync(); allowedPoemIds = new HashSet(ids); } else if (scopePoetId.HasValue) { var ids = await context.DiwanPoems.AsNoTracking() .Where(p => p.Cat.PoetId == scopePoetId.Value) .Select(p => p.Id) .ToListAsync(); allowedPoemIds = new HashSet(ids); } float[] queryVector = queryEmbedder.EmbedQuery(request.Query); // Fetch a larger candidate pool than the final k, so there's room for the // AI-summary re-ranking below to actually change the outcome — re-ranking a pool // that's already exactly k results wide can only reorder them, never let a // human-reviewed poem that was originally ranked k+1 overtake one that made the // cut only because its (unreviewed) summary happened to score marginally higher. int candidatePoolSize = Math.Min(k * CandidatePoolMultiplier, MaxTopK * CandidatePoolMultiplier); List<(int PoemId, float Score)> candidates = embeddingIndex.FindTopSimilar(queryVector, candidatePoolSize, allowedPoemIds); var candidateIds = candidates.Select(c => c.PoemId).ToList(); var poemsById = await context.DiwanPoems.AsNoTracking() .Where(p => candidateIds.Contains(p.Id)) .ToDictionaryAsync(p => p.Id); // Gentle re-rank: a poem whose summary is still AI-generated and un-reviewed // (still carries diwan-data's own "هوش مصنوعی:" prefix - removing it is part of // that project's human-review/edit workflow) gets a small score penalty before // final sorting. The DISPLAYED score stays the true, unpenalized cosine // similarity (Score below uses c.Score, not AdjustedScore) - the penalty is // purely an internal ordering nudge, not something that should make the shown // similarity number stop meaning what it says. float aiPenalty = _resources.AiSummaryScorePenalty; List<(int PoemId, float Score)> topMatches = candidates .Where(c => poemsById.ContainsKey(c.PoemId)) .Select(c => new { c.PoemId, c.Score, AdjustedScore = IsAiGeneratedSummary(poemsById[c.PoemId].PoemSummary) ? c.Score * aiPenalty : c.Score, }) .OrderByDescending(x => x.AdjustedScore) .Take(k) .Select(x => (x.PoemId, x.Score)) .ToList(); // computed once for the whole request, not per result - the query doesn't change // between results var queryKeywords = ExtractKeywords(request.Query); foreach (var match in topMatches) { // a poem id present in the embedding index but missing from the live DB // (e.g. deleted/unpublished since the embeddings were generated) is silently // skipped rather than failing the whole search for everyone else's results if (!poemsById.TryGetValue(match.PoemId, out var poem)) continue; // Fetch a bounded prefix of the poem's verses (not just the first // PreviewVerseCount) so SelectPreviewVerses has real material to search // through for a couplet that actually relates to the query, rather than // always showing the poem's opening lines regardless of relevance. Capped at // MaxVersesToScanForRelevance rather than the whole poem - covers the vast // majority of ghazals/qasides/robaiyat in full, and still a reasonable bound // for longer forms without pulling an unbounded number of rows per result. var candidateVerses = await context.DiwanVerses.AsNoTracking() .Where(v => v.PoemId == poem.Id) .OrderBy(v => v.VOrder) .Take(MaxVersesToScanForRelevance) .ToListAsync(); var verses = SelectPreviewVerses(candidateVerses, queryKeywords, PreviewVerseCount); response.Results.Add(new SemanticSearchResultDto { PoemId = poem.Id, Title = poem.Title, FullTitle = poem.FullTitle, FullUrl = poem.FullUrl, PoemSummary = poem.PoemSummary, Score = match.Score, Verses = verses.Select(v => new SemanticSearchVerseDto { VOrder = v.VOrder, Position = v.VersePosition.ToString(), Text = v.Text, }).ToList(), }); } // Best-effort logging, using the same context already open for this request - a // logging failure must never fail a real search, so any exception here is // swallowed and response.LogId simply stays null (ReportClickAsync already // no-ops safely if it's ever called with a null/0 LogId from a search that // couldn't be logged). try { var log = new SemanticSearchQueryLog { Query = request.Query, DateTimeUtc = DateTime.UtcNow, RequestedTopK = k, ResultCount = response.Results.Count, TopResultScore = response.Results.Count > 0 ? (float?)response.Results[0].Score : null, ScopeDetected = !string.IsNullOrEmpty(response.DetectedPoetName) || !string.IsNullOrEmpty(response.DetectedCategoryName), DetectedPoetName = response.DetectedPoetName, DetectedCategoryName = response.DetectedCategoryName, ScopeDetectionDisabled = request.DisableScopeDetection, }; context.SemanticSearchQueryLogs.Add(log); await context.SaveChangesAsync(); response.LogId = log.Id; } catch (Exception) { // logging is best-effort only - never worth failing a real search over } } return response; } public async Task ReportClickAsync(SemanticSearchClickDto click) { if (click == null || click.LogId <= 0) return; try { using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions())) { var log = await context.SemanticSearchQueryLogs.FindAsync(click.LogId); if (log != null) { log.ClickedPoemId = click.PoemId; log.ClickedResultRank = click.Rank; log.ClickedAtUtc = DateTime.UtcNow; await context.SaveChangesAsync(); } } } catch (Exception) { // click reporting is best-effort only - the click/navigation has already // happened regardless of whether this write succeeds, never worth surfacing an // error to the (fire-and-forget) caller for this } } /// /// diwan-data's own editing workflow requires this exact prefix be removed once a /// human has reviewed/edited a poem's summary — its continued presence is a direct, /// already-existing signal for "not yet human-reviewed," not something this project /// invented or has to infer. /// private static bool IsAiGeneratedSummary(string poemSummary) { return !string.IsNullOrEmpty(poemSummary) && poemSummary.StartsWith(AiGeneratedSummaryPrefix, StringComparison.Ordinal); } /// /// Common Persian function words stripped out before keyword-matching a query against /// verse text (see SelectPreviewVerses) — the kind of words that appear in nearly every /// query regardless of topic ("شعری در مورد ... پیدا کن") and would otherwise match /// almost any couplet in almost any poem, defeating the whole point of looking for a /// RELEVANT couplet rather than an arbitrary one. Not remotely exhaustive Persian /// stopword coverage — just the words that actually show up in how people phrase this /// kind of query, extended as real queries reveal gaps. /// private static readonly HashSet PersianStopWords = new HashSet { "شعر", "شعری", "شعرهای", "غزل", "غزلی", "قصیده", "رباعی", "مثنوی", "در", "مورد", "به", "از", "با", "برای", "را", "که", "یک", "این", "آن", "پیدا", "کن", "کنید", "کنم", "می‌خواهم", "میخواهم", "می‌خوام", "میخوام", "هست", "است", "بود", "تا", "یا", "و", "چه", "چگونه", "کدام", "چطور", "درباره", "دربارهٔ", "دربارۀ", "راجع", "بر", "روی", "های", "ها", "می", "خوب", "چیزی", "بگو", "بده", "نشان", "لطفا", "لطفاً", }; /// /// Splits the query on whitespace (deliberately NOT on ZWNJ — "بی‌وفایی" should survive /// as one token, not fracture into "بی" + "وفایی", where "بی" alone is a common enough /// prefix to false-positive-match all over the place), strips surrounding punctuation, /// drops stopwords and anything too short to be a meaningful keyword on its own. /// private static List ExtractKeywords(string query) { if (string.IsNullOrWhiteSpace(query)) return new List(); char[] punctuation = { '؟', '?', '.', '،', ',', '!', ':', ';', '«', '»', '"', '\'' }; return query .Split(new[] { ' ' }, StringSplitOptions.RemoveEmptyEntries) .Select(w => w.Trim(punctuation)) .Where(w => w.Length >= 2 && !PersianStopWords.Contains(w)) .Distinct() .ToList(); } /// /// Looks for the couplet (Right/Left verse pair) whose combined text contains the most /// query keywords, and returns it (plus, if there's room within previewVerseCount, the /// couplet immediately following it, for a little reading continuity rather than a /// single isolated pair). Falls back to the poem's opening verses — the previous, /// always-the-same-lines behavior — if no keyword appears anywhere in the scanned /// verses, or if there were no real keywords to search for at all (a query that was /// entirely stopwords, or empty after stripping them). /// /// Matches against CoupletSummary (when present) as well as the raw verse text — the /// summary is clean, modern-language prose, while the verses themselves are archaic and /// metaphorical and often won't literally contain a query's keywords even when the /// couplet is genuinely on-topic. CoupletSummary is only ever read from the FIRST verse /// of the pair (the "Right"/anchor position) — diwan-data stores it once per couplet /// there, never on the second ("Left") verse; a small number of "Left" rows do carry a /// stray value (a data anomaly, not a second legitimate copy), and this deliberately /// never reads it from that position, matching how the data is actually meant to be laid /// out rather than how a few rows happen to look. /// private static List SelectPreviewVerses(List allVerses, List keywords, int previewVerseCount) { if (keywords.Count > 0) { int bestScore = 0; int bestIndex = -1; for (int i = 0; i < allVerses.Count - 1; i++) { if (allVerses[i].VersePosition.ToString() != "Right" || allVerses[i + 1].VersePosition.ToString() != "Left") continue; string coupletText = allVerses[i].Text + " " + allVerses[i + 1].Text; if (!string.IsNullOrEmpty(allVerses[i].CoupletSummary)) { coupletText += " " + allVerses[i].CoupletSummary; } int score = keywords.Count(k => coupletText.Contains(k)); if (score > bestScore) { bestScore = score; bestIndex = i; } } if (bestIndex >= 0) { var result = new List { allVerses[bestIndex], allVerses[bestIndex + 1] }; int next = bestIndex + 2; while (result.Count < previewVerseCount && next < allVerses.Count) { result.Add(allVerses[next]); next++; } return result; } } // fallback: the poem's opening verses, same as the original behavior return allVerses.Take(previewVerseCount).ToList(); } /// /// Walks the whole category subtree rooted at rootCatId (breadth-first, level by level) /// and returns every category id in it, including rootCatId itself. Needed because a /// detected "book" category (شاهنامه, غزلیات, ...) is rarely where poems live directly — /// it's typically broken into many nested subcategories, with the actual poems several /// levels deeper. One query per depth level, not per node — a large book with many /// subcategories still only costs as many round-trips as the tree is deep, not how many /// nodes it has. /// private static async Task> GetDescendantCategoryIdsAsync(RMuseumDbContext context, int rootCatId) { var allCatIds = new HashSet { rootCatId }; var frontier = new List { rootCatId }; while (frontier.Count > 0) { var children = await context.DiwanCategories.AsNoTracking() .Where(c => c.ParentId.HasValue && frontier.Contains(c.ParentId.Value)) .Select(c => c.Id) .ToListAsync(); var newIds = children.Where(id => !allCatIds.Contains(id)).ToList(); foreach (var id in newIds) { allCatIds.Add(id); } frontier = newIds; } return allCatIds; } } }