using AngleSharp.Html.Parser; using Ganss.Xss; using Microsoft.EntityFrameworkCore; using Newtonsoft.Json; using RMuseum.DbContext; using RMuseum.Models.Divan; using RSecurityBackend.Models.Generic; using RSecurityBackend.Models.Generic.Db; using RSecurityBackend.Services.Implementation; using System; using System.Collections.Generic; using System.Data; using System.Linq; using System.Net.Http; using System.Text.RegularExpressions; using System.Threading.Tasks; namespace RMuseum.Services.Implementation { /// /// IDivanService implementation /// public partial class DivanService : IDivanService { private async Task _FindCategoryPoemsRhythmsInternal(int catId, bool retag, string rhythm) { using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions())) //this is long running job, so _context might be already been freed/collected by GC { LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context); var job = (await jobProgressServiceEF.NewJob($"FindCategoryPoemsRhythms Cat {catId}", "Query data")).Result; await _FindCategoryPoemsRhythmsMoreInternal(context, jobProgressServiceEF, job, catId, retag, rhythm); await jobProgressServiceEF.UpdateJob(job.Id, 0, "SubCats"); List catListId = new List(); await _populateCategoryChildren(context, catId, catListId); foreach (var subCatId in catListId) { await jobProgressServiceEF.UpdateJob(job.Id, subCatId, "SubCats"); await _FindCategoryPoemsRhythmsMoreInternal(context, jobProgressServiceEF, job, subCatId, retag, rhythm); } await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true); } } private async Task _FindCategoryPoemsRhythmsMoreInternal(RMuseumDbContext context, LongRunningJobProgressServiceEF jobProgressServiceEF, RLongRunningJobStatus job, int catId, bool retag, string rhythm) { try { var metres = await context.DivanMetres.OrderBy(m => m.Rhythm).AsNoTracking().ToArrayAsync(); var rhythms = metres.Select(m => m.Rhythm).ToArray(); DivanMetre preDeterminedMetre = string.IsNullOrEmpty(rhythm) ? null : metres.Where(m => m.Rhythm == rhythm).Single(); var poems = await context.DivanPoems.AsNoTracking().Where(p => p.CatId == catId).ToListAsync(); int i = 0; List> updateList = new List>(); using (HttpClient httpClient = new HttpClient()) { foreach (var poem in poems) { var sections = await context.DivanPoemSections.Where(p => p.PoemId == poem.Id).ToListAsync(); foreach (var section in sections.Where(s => s.DivanMetreRefSectionIndex == null).ToList()) { if (retag || section.DivanMetreId == null) { if (preDeterminedMetre == null) { var res = await _FindSectionRhythm(section, context, httpClient, rhythms); if (!string.IsNullOrEmpty(res.Result)) { section.DivanMetreId = metres.Where(m => m.Rhythm == res.Result).Single().Id; context.DivanPoemSections.Update(section); } } else { section.DivanMetreId = preDeterminedMetre.Id; context.DivanPoemSections.Update(section); } if (section.DivanMetreId != null) { var dependentSections = sections.Where(s => s.DivanMetreRefSectionIndex == section.Index).ToList(); foreach (var dsection in dependentSections) { dsection.DivanMetreId = section.DivanMetreId; context.DivanPoemSections.Update(dsection); } } if (section.DivanMetreId != null && !string.IsNullOrEmpty(section.RhymeLetters)) { if (!updateList.Any(u => u.Item1 == section.DivanMetreId && u.Item2 == section.RhymeLetters)) { updateList.Add(new Tuple((int)section.DivanMetreId, section.RhymeLetters)); } } if (section.DivanMetreId != null) { var dependentSections = sections.Where(s => s.DivanMetreRefSectionIndex == section.Index).ToList(); foreach (var dsection in dependentSections) { if (dsection.DivanMetreId != null && !string.IsNullOrEmpty(dsection.RhymeLetters)) { if (!updateList.Any(u => u.Item1 == dsection.DivanMetreId && u.Item2 == dsection.RhymeLetters)) { updateList.Add(new Tuple((int)dsection.DivanMetreId, dsection.RhymeLetters)); } } } } await jobProgressServiceEF.UpdateJob(job.Id, i++); } } } } foreach (var item in updateList) { await _UpdateRelatedSections(context, item.Item1, item.Item2, jobProgressServiceEF, job); } var subCats = await context.DivanCategories.AsNoTracking().Where(c => c.ParentId == catId).ToListAsync(); foreach (var subCat in subCats) { await jobProgressServiceEF.UpdateJob(job.Id, 0, $"FindCategoryPoemsRhythms Cat {subCat.Id}"); await _FindCategoryPoemsRhythmsMoreInternal(context, jobProgressServiceEF, job, subCat.Id, retag, rhythm); } } catch (Exception exp) { await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString()); } } /// /// find category poem rhymes /// /// /// /// /// public RServiceResult FindCategoryPoemsRhythms(int catId, bool retag, string rhythm = "") { _backgroundTaskQueue.QueueBackgroundWorkItem ( async token => { await _FindCategoryPoemsRhythmsInternal(catId, retag, rhythm); }); return new RServiceResult(true); } /// /// directly insert generated TOC /// /// /// /// /// public async Task> DirectInsertGeneratedTableOfContents(int catId, Guid userId, DivanTOC options) { return await _DirectInsertGeneratedTableOfContents(_context, catId, userId, options); } /// /// generate category TOC /// /// /// /// /// public async Task> GenerateTableOfContents(Guid userId, int catId, DivanTOC options) { return await _GenerateTableOfContents(_context, catId, options, userId); } private static readonly Regex _titleNumberRegex = new Regex(@"[۰-۹٠-٩0-9]+", RegexOptions.Compiled); private static readonly char[] _titleTrimChars = new[] { ' ', '\t', '\r', '\n', '\u200c', '\u200f', '\u200e', '-', '–', '—', ':', '،', ',', '.', '(', ')', '[', ']', '"', '\'', '«', '»' }; /// /// extracts the searchable part of a poem title, discarding pure boilerplate like /// "غزل شمارهٔ ۱" or "بخش ۴۷" which carries no information beyond the ordinal number, /// while keeping either the whole title (if it has no number at all, e.g. "طهمورث") /// or just the descriptive part after the number (e.g. "بخش ۴۷ - تشبیه کردن قرآن..." /// keeps "تشبیه کردن قرآن...") - returns "" when there is nothing worth indexing /// /// /// private static string ExtractSearchableTitlePart(string title) { if (string.IsNullOrWhiteSpace(title)) return ""; Match m = _titleNumberRegex.Match(title); if (!m.Success) return title.Trim(_titleTrimChars); return title.Substring(m.Index + m.Length).Trim(_titleTrimChars); } /// /// make plain text /// /// /// poem/section title - its searchable part (if any) is included so /// titles like "بخش ۴۷ - تشبیه کردن ..." are findable through search, while pure /// boilerplate like "غزل شمارهٔ ۱" is silently skipped /// private static string PreparePlainText(List verses, string title = null) { string plainText = ""; string searchableTitle = ExtractSearchableTitlePart(title); if (!string.IsNullOrEmpty(searchableTitle)) { plainText += $"{LanguageUtils.MakeTextSearchable(searchableTitle)}{Environment.NewLine}"; } foreach (DivanVerse verse in verses) { plainText += $"{LanguageUtils.MakeTextSearchable(verse.Text)}{Environment.NewLine}";//replace zwnj with space } return plainText.Trim(); } /// /// separate verses in poem.PlainText with Environment.NewLine instead of SPACE /// /// /// public RServiceResult RegerneratePoemsPlainText(int catId) { _backgroundTaskQueue.QueueBackgroundWorkItem ( async token => { using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions())) //this is long running job, so _context might be already been freed/collected by GC { LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context); var job = (await jobProgressServiceEF.NewJob($"RegerneratePoemsPlainText {catId}", "Query data")).Result; try { var poems = catId == 0 ? await context.DivanPoems.ToArrayAsync() : await context.DivanPoems.Where(p => p.CatId == catId).ToArrayAsync(); await jobProgressServiceEF.UpdateJob(job.Id, 0, $"Updating PlainText/Poem Html {catId}"); int percent = 0; for (int i = 0; i < poems.Length; i++) { if (i * 100 / poems.Length > percent) { percent++; await jobProgressServiceEF.UpdateJob(job.Id, percent); } var poem = poems[i]; var verses = await context.DivanVerses.AsNoTracking().Where(v => v.PoemId == poem.Id).OrderBy(v => v.VOrder).ToListAsync(); poem.PlainText = PreparePlainText(verses, poem.Title); poem.HtmlText = PrepareHtmlText(verses); } await jobProgressServiceEF.UpdateJob(job.Id, 0, $"Finalizing PlainText/Poem Html {catId}"); context.DivanPoems.UpdateRange(poems); await context.SaveChangesAsync(); await jobProgressServiceEF.UpdateJob(job.Id, 50, $"Updating pages HTML {catId}"); //the following line always gets timeout, so it is being replaced by a loop //await context.Database.ExecuteSqlRawAsync( // "UPDATE p SET p.HtmlText = (SELECT poem.HtmlText FROM DivanPoems poem WHERE poem.Id = p.Id) FROM DivanPages p WHERE p.DivanPageType = 3 "); foreach (var poem in poems) { var page = await context.DivanPages.Where(p => p.Id == poem.Id).SingleAsync(); page.HtmlText = poem.HtmlText; context.DivanPages.Update(page); } await context.SaveChangesAsync(); await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true); } catch (Exception exp) { await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString()); } } } ); return new RServiceResult(true); } /// /// examine site pages for broken links /// /// public RServiceResult HealthCheckContents() { _backgroundTaskQueue.QueueBackgroundWorkItem ( async token => { using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions())) //this is long running job, so _context might be already been freed/collected by GC { LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context); var job = (await jobProgressServiceEF.NewJob("HealthCheckContents", "Query data")).Result; try { var pages = await context.DivanPages.ToArrayAsync(); await jobProgressServiceEF.UpdateJob(job.Id, 0, $"Examining Pages"); var previousErrors = await context.DivanHealthCheckErrors.ToArrayAsync(); context.RemoveRange(previousErrors); await context.SaveChangesAsync(); int percent = 0; for (int i = 0; i < pages.Length; i++) { if (i * 100 / pages.Length > percent) { percent++; await jobProgressServiceEF.UpdateJob(job.Id, percent); } var hrefs = pages[i].HtmlText.Split(new[] { "href=\"" }, StringSplitOptions.RemoveEmptyEntries).Where(o => o.StartsWith("http")).Select(o => o.Substring(0, o.IndexOf("\""))); foreach (string url in hrefs) { if (url == "https://ganjoor.net" || url == "https://ganjoor.net/" || url.IndexOf("https://ganjoor.net/vazn/?") == 0 || url.IndexOf("https://ganjoor.net/simi/?v") == 0) continue; if (url.IndexOf("http://ganjoor.net") == 0) { context.DivanHealthCheckErrors.Add ( new DivanHealthCheckError() { ReferrerPageUrl = pages[i].FullUrl, TargetUrl = url, BrokenLink = false, MulipleTargets = false } ); await context.SaveChangesAsync(); } else if (url.IndexOf("https://ganjoor.net") == 0) { var testUrl = url.Substring("https://ganjoor.net".Length); if (testUrl[testUrl.Length - 1] == '/') testUrl = testUrl.Substring(0, testUrl.Length - 1); var pageCount = await context.DivanPages.Where(p => p.FullUrl == testUrl).CountAsync(); if (pageCount != 1) { context.DivanHealthCheckErrors.Add ( new DivanHealthCheckError() { ReferrerPageUrl = pages[i].FullUrl, TargetUrl = url, BrokenLink = pageCount == 0, MulipleTargets = pageCount != 0 } ); await context.SaveChangesAsync(); } } } } await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true); } catch (Exception exp) { await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString()); } } } ); return new RServiceResult(true); } /// /// extracts normalized plain text from (possibly malformed) HTML the same way a /// spec-compliant parser (and so also the sanitizer's own parser) sees it, so it can /// be compared before/after sanitizing to detect whether sanitizing dropped real text /// private static string _ExtractPlainText(string html) { if (string.IsNullOrWhiteSpace(html)) return ""; var parser = new HtmlParser(); using var document = parser.ParseDocument($"{html}"); string text = document.Body?.TextContent ?? ""; return Regex.Replace(text, @"\s+", " ").Trim(); } /// /// true if sanitizing the comment dropped a meaningful chunk of the user's actual /// text - not just markup/attributes, and not a harmless space lost when an inline /// tag gets unwrapped. This happens when invalid/unclosed markup causes real comment /// text to end up nested inside a tag the sanitizer correctly removes entirely /// (e.g. a stray/unclosed tag that swallows the rest of the comment as its "content", /// or an actually-disallowed tag such as script/style whose whole subtree is removed) /// private static bool _CommentSanitizationDroppedText(string originalPlainText, string sanitizedPlainText) { originalPlainText = (originalPlainText ?? "").Trim(); sanitizedPlainText = (sanitizedPlainText ?? "").Trim(); if (originalPlainText.Length == 0) return false; // nothing to lose if (sanitizedPlainText.Length == 0) return true; // the whole comment text vanished if (originalPlainText.Length - sanitizedPlainText.Length <= 2) return false; // negligible, e.g. a boundary space lost when unwrapping a tag return sanitizedPlainText.Length < originalPlainText.Length * 0.95; } /// /// shown to the user (via _BuildSanitizerDroppedTextError) when sanitizing had to drop /// real text. Deliberately says nothing about "tags", "<" or ">" - ordinary users /// don't know what those are and have usually never typed one themselves; almost always /// this happens because they pasted the text in from somewhere else (Word, a chat app, a /// web page) that silently carried invalid/broken formatting along with it. /// private const string _sanitizerTextDroppedMessage = "به‌نظر می‌رسد متنی که ارسال کرده‌اید از جای دیگری (مثلاً Word یا یک صفحهٔ وب) کپی و در اینجا پیست شده و قالب‌بندی پنهان و نامعتبری را با خود آورده است. به همین دلیل بخشی از متن هنگام پاک‌سازی حذف شد؛ بخش حذف‌شده در ادامه با رنگ قرمز و خط‌خورده به شما نشان داده می‌شود. لطفاً متن را در یک ویرایشگر متن ساده (مثل Notepad) پاک‌سازی کنید یا از نو تایپ کنید و سپس دوباره ارسال نمایید."; /// /// builds the error string returned to the client when sanitizing dropped real text. /// This is a JSON object encoded as a string (not a status-code/contract change) so it /// still flows through every existing "the API error is just a display string" code path /// unchanged, while DivanRazor can additionally recognize and unpack it to show the /// user exactly what got dropped (see SanitizerTextDroppedInfo on the DivanRazor side) /// private static string _BuildSanitizerDroppedTextError(string remainingPlainText) { return JsonConvert.SerializeObject(new { sanitizerTextDropped = true, message = _sanitizerTextDroppedMessage, remainingText = remainingPlainText ?? "" }); } /// /// sanitizes comment HTML; TextWasDropped is true when the sanitizing process removed /// a meaningful chunk of the user's actual text (see _CommentSanitizationDroppedText) - /// callers should not silently save Html in that case, but ask the user to fix their /// markup instead (RemainingPlainText/_BuildSanitizerDroppedTextError let them show what /// was dropped) /// private async Task<(string Html, bool TextWasDropped, string RemainingPlainText)> _ProcessCommentHtml(string commentText, RMuseumDbContext context) { string originalPlainText = _ExtractPlainText(commentText); // Use a proper HTML sanitizer var sanitizer = new HtmlSanitizer(); // Configure allowed tags sanitizer.AllowedTags.Clear(); sanitizer.AllowedTags.Add("p"); sanitizer.AllowedTags.Add("a"); sanitizer.AllowedTags.Add("br"); sanitizer.AllowedTags.Add("b"); sanitizer.AllowedTags.Add("i"); sanitizer.AllowedTags.Add("strong"); sanitizer.AllowedTags.Add("img"); sanitizer.AllowedTags.Add("span"); // Configure allowed attributes sanitizer.AllowedAttributes.Clear(); sanitizer.AllowedAttributes.Add("href"); sanitizer.AllowedAttributes.Add("src"); sanitizer.AllowedAttributes.Add("alt"); sanitizer.AllowedAttributes.Add("title"); sanitizer.AllowedAttributes.Add("rel"); // IMPORTANT: Remove all style attributes to prevent colored text and font changes sanitizer.AllowedAttributes.Remove("style"); // Disallow any CSS or style-related attributes sanitizer.AllowedCssProperties.Clear(); // Additional security: disallow data URIs and javascript: protocols sanitizer.AllowedSchemes.Clear(); sanitizer.AllowedSchemes.Add("http"); sanitizer.AllowedSchemes.Add("https"); sanitizer.AllowedSchemes.Add("mailto"); sanitizer.AllowedSchemes.Add("ftp"); // Sanitize the HTML string sanitizedHtml = sanitizer.Sanitize(commentText); // Detect whether sanitizing took real text down along with the invalid markup it // removed, before Linkify/internal-link processing below can itself change the // visible text (e.g. replacing a bare URL's text with a page title) in a way that // is not a loss. string sanitizedPlainText = _ExtractPlainText(sanitizedHtml); bool textWasDropped = _CommentSanitizationDroppedText(originalPlainText, sanitizedPlainText); // Process URLs (Linkify) and internal Divan links sanitizedHtml = await _ProcessUrls(sanitizedHtml, context); return (sanitizedHtml, textWasDropped, sanitizedPlainText); } private async Task _ProcessUrls(string html, RMuseumDbContext context) { // First, process any existing href links html = await _ProcessExistingLinks(html, context); // Then linkify any bare URLs that aren't already in links if (!html.Contains("href=\"") && html.Contains("http")) { html = _Linkify(html); } return html; } private async Task _ProcessExistingLinks(string html, RMuseumDbContext context) { // Use regex to find and process all href attributes var hrefRegex = new Regex(@"]*?\s+)?href=""([^""]*)""[^>]*>(.*?)", RegexOptions.IgnoreCase | RegexOptions.Singleline); var matches = hrefRegex.Matches(html); var result = html; var replacements = new List<(string old, string @new)>(); foreach (Match match in matches) { string url = match.Groups[1].Value; string linkText = match.Groups[2].Value; string newLink = await _ProcessSingleLink(url, linkText, context); // Store the replacement replacements.Add((match.Value, newLink)); } // Apply replacements from end to start to maintain indices foreach (var replacement in replacements.OrderByDescending(r => result.IndexOf(r.old))) { result = result.Replace(replacement.old, replacement.@new); } return result; } private async Task _ProcessSingleLink(string url, string linkText, RMuseumDbContext context) { // If link text is the same as URL (auto-generated link), try to improve it if (url == linkText || linkText.Trim() == url.Trim()) { // Process Divan internal links if (url.StartsWith("http://ganjoor.net") || url.StartsWith("https://ganjoor.net")) { string path = url.Replace("http://ganjoor.net", "").Replace("https://ganjoor.net", ""); int coupletNumber = -1; string cleanPath = path; // Extract couplet number if present if (path.Contains("#bn")) { int coupletStartIndex = path.IndexOf("#bn") + "#bn".Length; if (int.TryParse(path.Substring(coupletStartIndex), out coupletNumber)) { cleanPath = path.Substring(0, path.IndexOf("#bn")); } } // Remove trailing slash if (cleanPath.Length > 0 && cleanPath[cleanPath.Length - 1] == '/') cleanPath = cleanPath.Substring(0, cleanPath.Length - 1); var page = await context.DivanPages .AsNoTracking() .Where(p => p.FullUrl == cleanPath) .FirstOrDefaultAsync(); if (page != null) { string displayText = page.FullTitle; // Add couplet summary if applicable if (coupletNumber != -1) { string coupletSummary = await _GetCoupletSummary(page.Id, coupletNumber); if (!string.IsNullOrEmpty(coupletSummary)) { displayText = $"{page.FullTitle} » {coupletSummary}"; } } return $@"{displayText}"; } } else { // External link - use generic text return $@"پیوند به وبگاه بیرونی"; } } // If link text was manually provided, keep it but still validate the link return $@"{linkText}"; } private async Task _GetCoupletSummary(int poemId, int coupletNumber) { int coupletIndex = coupletNumber - 1; var verses = await _context.DivanVerses .Where(v => v.PoemId == poemId) .OrderBy(v => v.VOrder) .ToListAsync(); int cIndex = -1; for (int i = 0; i < verses.Count; i++) { if (verses[i].VersePosition != VersePosition.Left && verses[i].VersePosition != VersePosition.CenteredVerse2) cIndex++; if (cIndex == coupletIndex) { string coupletSummary = verses[i].Text; if (verses[i].VersePosition == VersePosition.Right && i < verses.Count - 1) { coupletSummary += $" {verses[i + 1].Text}"; } else if (verses[i].VersePosition == VersePosition.CenteredVerse1 && i < verses.Count - 1 && verses[i + 1].VersePosition == VersePosition.CenteredVerse2) { coupletSummary += $" {verses[i + 1].Text}"; } return _CutSummary(coupletSummary); } } return null; } private string _Linkify(string text) { // Improved Linkify with better URL detection var urlRegex = new Regex(@"(https?://[^\s<>""']+)", RegexOptions.IgnoreCase); return urlRegex.Replace(text, match => { string url = match.Value; return $@"{url}"; }); } /// /// examine comments for long links /// /// public RServiceResult FindAndFixLongUrlsInComments() { _backgroundTaskQueue.QueueBackgroundWorkItem ( async token => { using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions())) //this is long running job, so _context might be already been freed/collected by GC { LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context); var job = (await jobProgressServiceEF.NewJob("FindAndFixLongUrlsInComments", "Query data")).Result; try { var comments = //await context.DivanComments.Where(c => c.HtmlComment.Contains("href=")).ToArrayAsync(); await context.DivanComments.Where(c => c.HtmlComment.Contains("style")).ToArrayAsync(); await jobProgressServiceEF.UpdateJob(job.Id, 0, $"Examining {comments.Length} Comments"); int percent = 0; for (int i = 0; i < comments.Length; i++) { if (i * 100 / comments.Length > percent) { percent++; await jobProgressServiceEF.UpdateJob(job.Id, percent); } var comment = comments[i]; var processedComment = await _ProcessCommentHtml(comment.HtmlComment, context); if (processedComment.TextWasDropped) { // this is an unattended batch job re-sanitizing old comments, // there is no user here to ask to fix their markup - leave the // comment untouched rather than silently overwriting it with a // mutilated version continue; } if (processedComment.Html != comment.HtmlComment) { comment.HtmlComment = processedComment.Html; context.Update(comment); await context.SaveChangesAsync(); } } await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true); } catch (Exception exp) { await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString()); } } } ); return new RServiceResult(true); } /// /// start filling poems couplet indices /// /// public RServiceResult StartFillingPoemsCoupletIndices() { _backgroundTaskQueue.QueueBackgroundWorkItem ( async token => { using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions())) //this is long running job, so _context might be already been freed/collected by GC { LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context); var job = (await jobProgressServiceEF.NewJob("FillingPoemsCoupletIndices", "Query data")).Result; try { var poemIds = await context.DivanPoems.AsNoTracking().Select(p => p.Id).ToListAsync(); int percent = 0; for (int i = 0; i < poemIds.Count; i++) { if (i * 100 / poemIds.Count > percent) { percent++; await jobProgressServiceEF.UpdateJob(job.Id, percent); } await _FillPoemCoupletIndices(context, poemIds[i]); } await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true); } catch (Exception exp) { await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString()); } } } ); return new RServiceResult(true); } /// /// refill couplet indices /// /// /// public async Task> RefillCoupletIndicesAsync(int poemId) { try { await _FillPoemCoupletIndices(_context, poemId); await _context.SaveChangesAsync(); return new RServiceResult(false); } catch (Exception exp) { return new RServiceResult(false, exp.ToString()); } } /// /// fill section couplets count /// /// public RServiceResult StartFillingSectionCoupletCounts() { _backgroundTaskQueue.QueueBackgroundWorkItem ( async token => { using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions())) //this is long running job, so _context might be already been freed/collected by GC { LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context); var job = (await jobProgressServiceEF.NewJob("FillingSectionCoupletCounts", "Query data")).Result; try { var sectionIds = await context.DivanPoemSections.AsNoTracking().Select(p => p.Id).ToListAsync(); int percent = 0; for (int i = 0; i < sectionIds.Count; i++) { if (i * 100 / sectionIds.Count > percent) { percent++; await jobProgressServiceEF.UpdateJob(job.Id, percent); } await _FillSectionCoupletCountsAsync(context, sectionIds[i]); } await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true); } catch (Exception exp) { await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString()); } } } ); return new RServiceResult(true); } private async Task _FillSectionCoupletCountsAsync(RMuseumDbContext context, int sectionId) { var section = await context.DivanPoemSections.Where(s => s.Id == sectionId).SingleAsync(); section.CoupletsCount = await context.DivanVerses.Where( v => v.PoemId == section.PoemId && ( (section.VerseType == VersePoemSectionType.First && v.SectionIndex1 == section.Index) || (section.VerseType == VersePoemSectionType.Second && v.SectionIndex2 == section.Index) || (section.VerseType == VersePoemSectionType.Third && v.SectionIndex3 == section.Index) || (section.VerseType == VersePoemSectionType.Forth && v.SectionIndex4 == section.Index)) && (v.VersePosition == VersePosition.Left || v.VersePosition == VersePosition.CenteredVerse1) ).CountAsync(); context.Update(section); } /// /// regenerate poem full titles to fix an old bug /// /// public RServiceResult RegeneratePoemsFullTitles() { _backgroundTaskQueue.QueueBackgroundWorkItem ( async token => { using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions())) //this is long running job, so _context might be already been freed/collected by GC { LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context); var job = (await jobProgressServiceEF.NewJob("RegenratePoemsFullTitles", "Query data")).Result; int percent = 0; try { var poemIds = await context.DivanPoems.AsNoTracking().Select(p => p.Id).ToListAsync(); for (int i = 0; i < poemIds.Count; i++) { if (i * 100 / poemIds.Count > percent) { percent++; await jobProgressServiceEF.UpdateJob(job.Id, percent); } var poem = await context.DivanPoems.Where(p => p.Id == poemIds[i]).SingleOrDefaultAsync(); var catPage = await context.DivanPages.AsNoTracking().Where(p => p.DivanPageType == DivanPageType.CatPage && p.CatId == poem.CatId).SingleOrDefaultAsync(); if (catPage == null) { catPage = await context.DivanPages.AsNoTracking().Where(p => p.DivanPageType == DivanPageType.PoetPage && p.CatId == poem.CatId).SingleOrDefaultAsync(); } var title = $"{catPage.FullTitle} » {poem.Title}"; if (title != poem.FullTitle) { poem.FullTitle = title; context.Update(poem); } var page = await context.DivanPages.Where(p => p.Id == poem.Id).SingleOrDefaultAsync(); if (title != page.FullTitle) { page.FullTitle = title; context.Update(page); } } await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true); } catch (Exception exp) { await jobProgressServiceEF.UpdateJob(job.Id, percent, "", false, exp.ToString()); } } } ); return new RServiceResult(true); } /// /// start finding rhymes for single couplets /// /// public RServiceResult StartFindingRhymesForSingleCouplets() { _backgroundTaskQueue.QueueBackgroundWorkItem ( async token => { using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions())) //this is long running job, so _context might be already been freed/collected by GC { LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context); var job = (await jobProgressServiceEF.NewJob("FindingRhymesForSingleCouplets", "Query data")).Result; try { var sections = await context.DivanPoemSections.AsNoTracking().Include(s => s.DivanMetre).Where(s => s.SectionType == PoemSectionType.WholePoem && s.DivanMetre != null && (s.RhymeLetters.Length >= 30 || s.RhymeLetters.Length < 2)).ToListAsync(); int percent = 0; for (int i = 0; i < sections.Count; i++) { if (i * 100 / sections.Count > percent) { percent++; await jobProgressServiceEF.UpdateJob(job.Id, percent); } var res = await _FindSectionRhyme(context, sections[i].Id); if (string.IsNullOrEmpty(res.ExceptionString)) { if (!string.IsNullOrEmpty(res.Result.Rhyme) && res.Result.Rhyme != sections[i].RhymeLetters) { var sectionTracked = await context.DivanPoemSections.Where(s => s.Id == sections[i].Id).SingleAsync(); var oldRhyme = sectionTracked.RhymeLetters; sectionTracked.RhymeLetters = res.Result.Rhyme; context.Update(sectionTracked); await context.SaveChangesAsync(); await _UpdateRelatedSections(context, (int)sectionTracked.DivanMetreId, oldRhyme, jobProgressServiceEF, job, percent); await _UpdateRelatedSections(context, (int)sectionTracked.DivanMetreId, res.Result.Rhyme, jobProgressServiceEF, job, percent); } } } await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true); } catch (Exception exp) { await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString()); } } } ); return new RServiceResult(true); } } }