1007 lines
47 KiB
C#
1007 lines
47 KiB
C#
using AngleSharp.Html.Parser;
|
||
using Ganss.Xss;
|
||
using Microsoft.EntityFrameworkCore;
|
||
using Newtonsoft.Json;
|
||
using RMuseum.DbContext;
|
||
using RMuseum.Models.Ganjoor;
|
||
using RSecurityBackend.Models.Generic;
|
||
using RSecurityBackend.Models.Generic.Db;
|
||
using RSecurityBackend.Services.Implementation;
|
||
using System;
|
||
using System.Collections.Generic;
|
||
using System.Data;
|
||
using System.Linq;
|
||
using System.Net.Http;
|
||
using System.Text.RegularExpressions;
|
||
using System.Threading.Tasks;
|
||
|
||
namespace RMuseum.Services.Implementation
|
||
{
|
||
/// <summary>
|
||
/// IGanjoorService implementation
|
||
/// </summary>
|
||
public partial class GanjoorService : IGanjoorService
|
||
{
|
||
private async Task _FindCategoryPoemsRhythmsInternal(int catId, bool retag, string rhythm)
|
||
{
|
||
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>())) //this is long running job, so _context might be already been freed/collected by GC
|
||
{
|
||
LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context);
|
||
var job = (await jobProgressServiceEF.NewJob($"FindCategoryPoemsRhythms Cat {catId}", "Query data")).Result;
|
||
await _FindCategoryPoemsRhythmsMoreInternal(context, jobProgressServiceEF, job, catId, retag, rhythm);
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 0, "SubCats");
|
||
List<int> catListId = new List<int>();
|
||
await _populateCategoryChildren(context, catId, catListId);
|
||
foreach (var subCatId in catListId)
|
||
{
|
||
await jobProgressServiceEF.UpdateJob(job.Id, subCatId, "SubCats");
|
||
await _FindCategoryPoemsRhythmsMoreInternal(context, jobProgressServiceEF, job, subCatId, retag, rhythm);
|
||
}
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true);
|
||
}
|
||
}
|
||
|
||
private async Task _FindCategoryPoemsRhythmsMoreInternal(RMuseumDbContext context, LongRunningJobProgressServiceEF jobProgressServiceEF, RLongRunningJobStatus job, int catId, bool retag, string rhythm)
|
||
{
|
||
try
|
||
{
|
||
var metres = await context.GanjoorMetres.OrderBy(m => m.Rhythm).AsNoTracking().ToArrayAsync();
|
||
var rhythms = metres.Select(m => m.Rhythm).ToArray();
|
||
|
||
GanjoorMetre preDeterminedMetre = string.IsNullOrEmpty(rhythm) ? null : metres.Where(m => m.Rhythm == rhythm).Single();
|
||
|
||
var poems = await context.GanjoorPoems.AsNoTracking().Where(p => p.CatId == catId).ToListAsync();
|
||
|
||
int i = 0;
|
||
List<Tuple<int, string>> updateList = new List<Tuple<int, string>>();
|
||
using (HttpClient httpClient = new HttpClient())
|
||
{
|
||
|
||
foreach (var poem in poems)
|
||
{
|
||
|
||
var sections = await context.GanjoorPoemSections.Where(p => p.PoemId == poem.Id).ToListAsync();
|
||
foreach (var section in sections.Where(s => s.GanjoorMetreRefSectionIndex == null).ToList())
|
||
{
|
||
if (retag || section.GanjoorMetreId == null)
|
||
{
|
||
|
||
if (preDeterminedMetre == null)
|
||
{
|
||
var res = await _FindSectionRhythm(section, context, httpClient, rhythms);
|
||
if (!string.IsNullOrEmpty(res.Result))
|
||
{
|
||
section.GanjoorMetreId = metres.Where(m => m.Rhythm == res.Result).Single().Id;
|
||
context.GanjoorPoemSections.Update(section);
|
||
}
|
||
}
|
||
else
|
||
{
|
||
section.GanjoorMetreId = preDeterminedMetre.Id;
|
||
context.GanjoorPoemSections.Update(section);
|
||
}
|
||
|
||
if (section.GanjoorMetreId != null)
|
||
{
|
||
var dependentSections = sections.Where(s => s.GanjoorMetreRefSectionIndex == section.Index).ToList();
|
||
foreach (var dsection in dependentSections)
|
||
{
|
||
dsection.GanjoorMetreId = section.GanjoorMetreId;
|
||
context.GanjoorPoemSections.Update(dsection);
|
||
}
|
||
}
|
||
|
||
if (section.GanjoorMetreId != null && !string.IsNullOrEmpty(section.RhymeLetters))
|
||
{
|
||
if (!updateList.Any(u => u.Item1 == section.GanjoorMetreId && u.Item2 == section.RhymeLetters))
|
||
{
|
||
updateList.Add(new Tuple<int, string>((int)section.GanjoorMetreId, section.RhymeLetters));
|
||
}
|
||
|
||
}
|
||
|
||
if (section.GanjoorMetreId != null)
|
||
{
|
||
var dependentSections = sections.Where(s => s.GanjoorMetreRefSectionIndex == section.Index).ToList();
|
||
foreach (var dsection in dependentSections)
|
||
{
|
||
if (dsection.GanjoorMetreId != null && !string.IsNullOrEmpty(dsection.RhymeLetters))
|
||
{
|
||
if (!updateList.Any(u => u.Item1 == dsection.GanjoorMetreId && u.Item2 == dsection.RhymeLetters))
|
||
{
|
||
updateList.Add(new Tuple<int, string>((int)dsection.GanjoorMetreId, dsection.RhymeLetters));
|
||
}
|
||
|
||
}
|
||
}
|
||
}
|
||
|
||
await jobProgressServiceEF.UpdateJob(job.Id, i++);
|
||
}
|
||
}
|
||
|
||
|
||
}
|
||
}
|
||
|
||
foreach (var item in updateList)
|
||
{
|
||
await _UpdateRelatedSections(context, item.Item1, item.Item2, jobProgressServiceEF, job);
|
||
}
|
||
|
||
var subCats = await context.GanjoorCategories.AsNoTracking().Where(c => c.ParentId == catId).ToListAsync();
|
||
foreach (var subCat in subCats)
|
||
{
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 0, $"FindCategoryPoemsRhythms Cat {subCat.Id}");
|
||
await _FindCategoryPoemsRhythmsMoreInternal(context, jobProgressServiceEF, job, subCat.Id, retag, rhythm);
|
||
}
|
||
|
||
|
||
}
|
||
catch (Exception exp)
|
||
{
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString());
|
||
}
|
||
}
|
||
|
||
/// <summary>
|
||
/// find category poem rhymes
|
||
/// </summary>
|
||
/// <param name="catId"></param>
|
||
/// <param name="retag"></param>
|
||
/// <param name="rhythm"></param>
|
||
/// <returns></returns>
|
||
public RServiceResult<bool> FindCategoryPoemsRhythms(int catId, bool retag, string rhythm = "")
|
||
{
|
||
_backgroundTaskQueue.QueueBackgroundWorkItem
|
||
(
|
||
async token =>
|
||
{
|
||
await _FindCategoryPoemsRhythmsInternal(catId, retag, rhythm);
|
||
});
|
||
|
||
return new RServiceResult<bool>(true);
|
||
}
|
||
|
||
|
||
|
||
/// <summary>
|
||
/// directly insert generated TOC
|
||
/// </summary>
|
||
/// <param name="catId"></param>
|
||
/// <param name="userId"></param>
|
||
/// <param name="options"></param>
|
||
/// <returns></returns>
|
||
public async Task<RServiceResult<bool>> DirectInsertGeneratedTableOfContents(int catId, Guid userId, GanjoorTOC options)
|
||
{
|
||
return await _DirectInsertGeneratedTableOfContents(_context, catId, userId, options);
|
||
}
|
||
|
||
/// <summary>
|
||
/// generate category TOC
|
||
/// </summary>
|
||
/// <param name="userId"></param>
|
||
/// <param name="catId"></param>
|
||
/// <param name="options"></param>
|
||
/// <returns></returns>
|
||
public async Task<RServiceResult<string>> GenerateTableOfContents(Guid userId, int catId, GanjoorTOC options)
|
||
{
|
||
return await _GenerateTableOfContents(_context, catId, options, userId);
|
||
}
|
||
|
||
|
||
private static readonly Regex _titleNumberRegex = new Regex(@"[۰-۹٠-٩0-9]+", RegexOptions.Compiled);
|
||
private static readonly char[] _titleTrimChars = new[] { ' ', '\t', '\r', '\n', '\u200c', '\u200f', '\u200e', '-', '–', '—', ':', '،', ',', '.', '(', ')', '[', ']', '"', '\'', '«', '»' };
|
||
|
||
/// <summary>
|
||
/// extracts the searchable part of a poem title, discarding pure boilerplate like
|
||
/// "غزل شمارهٔ ۱" or "بخش ۴۷" which carries no information beyond the ordinal number,
|
||
/// while keeping either the whole title (if it has no number at all, e.g. "طهمورث")
|
||
/// or just the descriptive part after the number (e.g. "بخش ۴۷ - تشبیه کردن قرآن..."
|
||
/// keeps "تشبیه کردن قرآن...") - returns "" when there is nothing worth indexing
|
||
/// </summary>
|
||
/// <param name="title"></param>
|
||
/// <returns></returns>
|
||
private static string ExtractSearchableTitlePart(string title)
|
||
{
|
||
if (string.IsNullOrWhiteSpace(title))
|
||
return "";
|
||
|
||
Match m = _titleNumberRegex.Match(title);
|
||
if (!m.Success)
|
||
return title.Trim(_titleTrimChars);
|
||
|
||
return title.Substring(m.Index + m.Length).Trim(_titleTrimChars);
|
||
}
|
||
|
||
/// <summary>
|
||
/// make plain text
|
||
/// </summary>
|
||
/// <param name="verses"></param>
|
||
/// <param name="title">poem/section title - its searchable part (if any) is included so
|
||
/// titles like "بخش ۴۷ - تشبیه کردن ..." are findable through search, while pure
|
||
/// boilerplate like "غزل شمارهٔ ۱" is silently skipped</param>
|
||
/// <returns></returns>
|
||
private static string PreparePlainText(List<GanjoorVerse> verses, string title = null)
|
||
{
|
||
string plainText = "";
|
||
string searchableTitle = ExtractSearchableTitlePart(title);
|
||
if (!string.IsNullOrEmpty(searchableTitle))
|
||
{
|
||
plainText += $"{LanguageUtils.MakeTextSearchable(searchableTitle)}{Environment.NewLine}";
|
||
}
|
||
foreach (GanjoorVerse verse in verses)
|
||
{
|
||
plainText += $"{LanguageUtils.MakeTextSearchable(verse.Text)}{Environment.NewLine}";//replace zwnj with space
|
||
}
|
||
return plainText.Trim();
|
||
}
|
||
|
||
/// <summary>
|
||
/// separate verses in poem.PlainText with Environment.NewLine instead of SPACE
|
||
/// </summary>
|
||
/// <param name="catId"></param>
|
||
/// <returns></returns>
|
||
public RServiceResult<bool> RegerneratePoemsPlainText(int catId)
|
||
{
|
||
_backgroundTaskQueue.QueueBackgroundWorkItem
|
||
(
|
||
async token =>
|
||
{
|
||
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>())) //this is long running job, so _context might be already been freed/collected by GC
|
||
{
|
||
LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context);
|
||
var job = (await jobProgressServiceEF.NewJob($"RegerneratePoemsPlainText {catId}", "Query data")).Result;
|
||
|
||
try
|
||
{
|
||
var poems = catId == 0 ? await context.GanjoorPoems.ToArrayAsync() : await context.GanjoorPoems.Where(p => p.CatId == catId).ToArrayAsync();
|
||
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 0, $"Updating PlainText/Poem Html {catId}");
|
||
|
||
int percent = 0;
|
||
for (int i = 0; i < poems.Length; i++)
|
||
{
|
||
if (i * 100 / poems.Length > percent)
|
||
{
|
||
percent++;
|
||
await jobProgressServiceEF.UpdateJob(job.Id, percent);
|
||
}
|
||
|
||
var poem = poems[i];
|
||
|
||
var verses = await context.GanjoorVerses.AsNoTracking().Where(v => v.PoemId == poem.Id).OrderBy(v => v.VOrder).ToListAsync();
|
||
|
||
poem.PlainText = PreparePlainText(verses, poem.Title);
|
||
poem.HtmlText = PrepareHtmlText(verses);
|
||
}
|
||
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 0, $"Finalizing PlainText/Poem Html {catId}");
|
||
|
||
context.GanjoorPoems.UpdateRange(poems);
|
||
|
||
await context.SaveChangesAsync();
|
||
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 50, $"Updating pages HTML {catId}");
|
||
|
||
//the following line always gets timeout, so it is being replaced by a loop
|
||
//await context.Database.ExecuteSqlRawAsync(
|
||
// "UPDATE p SET p.HtmlText = (SELECT poem.HtmlText FROM GanjoorPoems poem WHERE poem.Id = p.Id) FROM GanjoorPages p WHERE p.GanjoorPageType = 3 ");
|
||
|
||
foreach (var poem in poems)
|
||
{
|
||
var page = await context.GanjoorPages.Where(p => p.Id == poem.Id).SingleAsync();
|
||
page.HtmlText = poem.HtmlText;
|
||
context.GanjoorPages.Update(page);
|
||
}
|
||
await context.SaveChangesAsync();
|
||
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true);
|
||
}
|
||
catch (Exception exp)
|
||
{
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString());
|
||
}
|
||
|
||
}
|
||
}
|
||
);
|
||
|
||
return new RServiceResult<bool>(true);
|
||
}
|
||
|
||
/// <summary>
|
||
/// examine site pages for broken links
|
||
/// </summary>
|
||
/// <returns></returns>
|
||
public RServiceResult<bool> HealthCheckContents()
|
||
{
|
||
_backgroundTaskQueue.QueueBackgroundWorkItem
|
||
(
|
||
async token =>
|
||
{
|
||
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>())) //this is long running job, so _context might be already been freed/collected by GC
|
||
{
|
||
LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context);
|
||
var job = (await jobProgressServiceEF.NewJob("HealthCheckContents", "Query data")).Result;
|
||
|
||
try
|
||
{
|
||
var pages = await context.GanjoorPages.ToArrayAsync();
|
||
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 0, $"Examining Pages");
|
||
|
||
var previousErrors = await context.GanjoorHealthCheckErrors.ToArrayAsync();
|
||
context.RemoveRange(previousErrors);
|
||
await context.SaveChangesAsync();
|
||
int percent = 0;
|
||
for (int i = 0; i < pages.Length; i++)
|
||
{
|
||
if (i * 100 / pages.Length > percent)
|
||
{
|
||
percent++;
|
||
await jobProgressServiceEF.UpdateJob(job.Id, percent);
|
||
}
|
||
|
||
var hrefs = pages[i].HtmlText.Split(new[] { "href=\"" }, StringSplitOptions.RemoveEmptyEntries).Where(o => o.StartsWith("http")).Select(o => o.Substring(0, o.IndexOf("\"")));
|
||
|
||
foreach (string url in hrefs)
|
||
{
|
||
if (url == "https://ganjoor.net" || url == "https://ganjoor.net/" || url.IndexOf("https://ganjoor.net/vazn/?") == 0 || url.IndexOf("https://ganjoor.net/simi/?v") == 0)
|
||
continue;
|
||
if (url.IndexOf("http://ganjoor.net") == 0)
|
||
{
|
||
context.GanjoorHealthCheckErrors.Add
|
||
(
|
||
new GanjoorHealthCheckError()
|
||
{
|
||
ReferrerPageUrl = pages[i].FullUrl,
|
||
TargetUrl = url,
|
||
BrokenLink = false,
|
||
MulipleTargets = false
|
||
}
|
||
);
|
||
|
||
await context.SaveChangesAsync();
|
||
}
|
||
else
|
||
if (url.IndexOf("https://ganjoor.net") == 0)
|
||
{
|
||
var testUrl = url.Substring("https://ganjoor.net".Length);
|
||
if (testUrl[testUrl.Length - 1] == '/')
|
||
testUrl = testUrl.Substring(0, testUrl.Length - 1);
|
||
var pageCount = await context.GanjoorPages.Where(p => p.FullUrl == testUrl).CountAsync();
|
||
if (pageCount != 1)
|
||
{
|
||
context.GanjoorHealthCheckErrors.Add
|
||
(
|
||
new GanjoorHealthCheckError()
|
||
{
|
||
ReferrerPageUrl = pages[i].FullUrl,
|
||
TargetUrl = url,
|
||
BrokenLink = pageCount == 0,
|
||
MulipleTargets = pageCount != 0
|
||
}
|
||
);
|
||
|
||
await context.SaveChangesAsync();
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true);
|
||
}
|
||
catch (Exception exp)
|
||
{
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString());
|
||
}
|
||
|
||
}
|
||
}
|
||
);
|
||
|
||
return new RServiceResult<bool>(true);
|
||
}
|
||
|
||
/// <summary>
|
||
/// extracts normalized plain text from (possibly malformed) HTML the same way a
|
||
/// spec-compliant parser (and so also the sanitizer's own parser) sees it, so it can
|
||
/// be compared before/after sanitizing to detect whether sanitizing dropped real text
|
||
/// </summary>
|
||
private static string _ExtractPlainText(string html)
|
||
{
|
||
if (string.IsNullOrWhiteSpace(html))
|
||
return "";
|
||
|
||
var parser = new HtmlParser();
|
||
using var document = parser.ParseDocument($"<body>{html}</body>");
|
||
string text = document.Body?.TextContent ?? "";
|
||
return Regex.Replace(text, @"\s+", " ").Trim();
|
||
}
|
||
|
||
/// <summary>
|
||
/// true if sanitizing the comment dropped a meaningful chunk of the user's actual
|
||
/// text - not just markup/attributes, and not a harmless space lost when an inline
|
||
/// tag gets unwrapped. This happens when invalid/unclosed markup causes real comment
|
||
/// text to end up nested inside a tag the sanitizer correctly removes entirely
|
||
/// (e.g. a stray/unclosed tag that swallows the rest of the comment as its "content",
|
||
/// or an actually-disallowed tag such as script/style whose whole subtree is removed)
|
||
/// </summary>
|
||
private static bool _CommentSanitizationDroppedText(string originalPlainText, string sanitizedPlainText)
|
||
{
|
||
originalPlainText = (originalPlainText ?? "").Trim();
|
||
sanitizedPlainText = (sanitizedPlainText ?? "").Trim();
|
||
|
||
if (originalPlainText.Length == 0)
|
||
return false; // nothing to lose
|
||
|
||
if (sanitizedPlainText.Length == 0)
|
||
return true; // the whole comment text vanished
|
||
|
||
if (originalPlainText.Length - sanitizedPlainText.Length <= 2)
|
||
return false; // negligible, e.g. a boundary space lost when unwrapping a tag
|
||
|
||
return sanitizedPlainText.Length < originalPlainText.Length * 0.95;
|
||
}
|
||
|
||
/// <summary>
|
||
/// shown to the user (via _BuildSanitizerDroppedTextError) when sanitizing had to drop
|
||
/// real text. Deliberately says nothing about "tags", "<" or ">" - ordinary users
|
||
/// don't know what those are and have usually never typed one themselves; almost always
|
||
/// this happens because they pasted the text in from somewhere else (Word, a chat app, a
|
||
/// web page) that silently carried invalid/broken formatting along with it.
|
||
/// </summary>
|
||
private const string _sanitizerTextDroppedMessage =
|
||
"بهنظر میرسد متنی که ارسال کردهاید از جای دیگری (مثلاً Word یا یک صفحهٔ وب) کپی و در اینجا پیست شده و قالببندی پنهان و نامعتبری را با خود آورده است. به همین دلیل بخشی از متن هنگام پاکسازی حذف شد؛ بخش حذفشده در ادامه با رنگ قرمز و خطخورده به شما نشان داده میشود. لطفاً متن را در یک ویرایشگر متن ساده (مثل Notepad) پاکسازی کنید یا از نو تایپ کنید و سپس دوباره ارسال نمایید.";
|
||
|
||
/// <summary>
|
||
/// builds the error string returned to the client when sanitizing dropped real text.
|
||
/// This is a JSON object encoded as a string (not a status-code/contract change) so it
|
||
/// still flows through every existing "the API error is just a display string" code path
|
||
/// unchanged, while GanjooRazor can additionally recognize and unpack it to show the
|
||
/// user exactly what got dropped (see SanitizerTextDroppedInfo on the GanjooRazor side)
|
||
/// </summary>
|
||
private static string _BuildSanitizerDroppedTextError(string remainingPlainText)
|
||
{
|
||
return JsonConvert.SerializeObject(new
|
||
{
|
||
sanitizerTextDropped = true,
|
||
message = _sanitizerTextDroppedMessage,
|
||
remainingText = remainingPlainText ?? ""
|
||
});
|
||
}
|
||
|
||
/// <summary>
|
||
/// sanitizes comment HTML; TextWasDropped is true when the sanitizing process removed
|
||
/// a meaningful chunk of the user's actual text (see _CommentSanitizationDroppedText) -
|
||
/// callers should not silently save Html in that case, but ask the user to fix their
|
||
/// markup instead (RemainingPlainText/_BuildSanitizerDroppedTextError let them show what
|
||
/// was dropped)
|
||
/// </summary>
|
||
private async Task<(string Html, bool TextWasDropped, string RemainingPlainText)> _ProcessCommentHtml(string commentText, RMuseumDbContext context)
|
||
{
|
||
string originalPlainText = _ExtractPlainText(commentText);
|
||
|
||
// Use a proper HTML sanitizer
|
||
var sanitizer = new HtmlSanitizer();
|
||
|
||
// Configure allowed tags
|
||
sanitizer.AllowedTags.Clear();
|
||
sanitizer.AllowedTags.Add("p");
|
||
sanitizer.AllowedTags.Add("a");
|
||
sanitizer.AllowedTags.Add("br");
|
||
sanitizer.AllowedTags.Add("b");
|
||
sanitizer.AllowedTags.Add("i");
|
||
sanitizer.AllowedTags.Add("strong");
|
||
sanitizer.AllowedTags.Add("img");
|
||
sanitizer.AllowedTags.Add("span");
|
||
|
||
// Configure allowed attributes
|
||
sanitizer.AllowedAttributes.Clear();
|
||
sanitizer.AllowedAttributes.Add("href");
|
||
sanitizer.AllowedAttributes.Add("src");
|
||
sanitizer.AllowedAttributes.Add("alt");
|
||
sanitizer.AllowedAttributes.Add("title");
|
||
sanitizer.AllowedAttributes.Add("rel");
|
||
|
||
// IMPORTANT: Remove all style attributes to prevent colored text and font changes
|
||
sanitizer.AllowedAttributes.Remove("style");
|
||
|
||
// Disallow any CSS or style-related attributes
|
||
sanitizer.AllowedCssProperties.Clear();
|
||
|
||
// Additional security: disallow data URIs and javascript: protocols
|
||
sanitizer.AllowedSchemes.Clear();
|
||
sanitizer.AllowedSchemes.Add("http");
|
||
sanitizer.AllowedSchemes.Add("https");
|
||
sanitizer.AllowedSchemes.Add("mailto");
|
||
sanitizer.AllowedSchemes.Add("ftp");
|
||
|
||
// Sanitize the HTML
|
||
string sanitizedHtml = sanitizer.Sanitize(commentText);
|
||
|
||
// Detect whether sanitizing took real text down along with the invalid markup it
|
||
// removed, before Linkify/internal-link processing below can itself change the
|
||
// visible text (e.g. replacing a bare URL's text with a page title) in a way that
|
||
// is not a loss.
|
||
string sanitizedPlainText = _ExtractPlainText(sanitizedHtml);
|
||
bool textWasDropped = _CommentSanitizationDroppedText(originalPlainText, sanitizedPlainText);
|
||
|
||
// Process URLs (Linkify) and internal Ganjoor links
|
||
sanitizedHtml = await _ProcessUrls(sanitizedHtml, context);
|
||
|
||
return (sanitizedHtml, textWasDropped, sanitizedPlainText);
|
||
}
|
||
|
||
private async Task<string> _ProcessUrls(string html, RMuseumDbContext context)
|
||
{
|
||
// First, process any existing href links
|
||
html = await _ProcessExistingLinks(html, context);
|
||
|
||
// Then linkify any bare URLs that aren't already in links
|
||
if (!html.Contains("href=\"") && html.Contains("http"))
|
||
{
|
||
html = _Linkify(html);
|
||
}
|
||
|
||
return html;
|
||
}
|
||
|
||
private async Task<string> _ProcessExistingLinks(string html, RMuseumDbContext context)
|
||
{
|
||
// Use regex to find and process all href attributes
|
||
var hrefRegex = new Regex(@"<a\s+(?:[^>]*?\s+)?href=""([^""]*)""[^>]*>(.*?)</a>", RegexOptions.IgnoreCase | RegexOptions.Singleline);
|
||
|
||
var matches = hrefRegex.Matches(html);
|
||
var result = html;
|
||
var replacements = new List<(string old, string @new)>();
|
||
|
||
foreach (Match match in matches)
|
||
{
|
||
string url = match.Groups[1].Value;
|
||
string linkText = match.Groups[2].Value;
|
||
string newLink = await _ProcessSingleLink(url, linkText, context);
|
||
|
||
// Store the replacement
|
||
replacements.Add((match.Value, newLink));
|
||
}
|
||
|
||
// Apply replacements from end to start to maintain indices
|
||
foreach (var replacement in replacements.OrderByDescending(r => result.IndexOf(r.old)))
|
||
{
|
||
result = result.Replace(replacement.old, replacement.@new);
|
||
}
|
||
|
||
return result;
|
||
}
|
||
|
||
private async Task<string> _ProcessSingleLink(string url, string linkText, RMuseumDbContext context)
|
||
{
|
||
// If link text is the same as URL (auto-generated link), try to improve it
|
||
if (url == linkText || linkText.Trim() == url.Trim())
|
||
{
|
||
// Process Ganjoor internal links
|
||
if (url.StartsWith("http://ganjoor.net") || url.StartsWith("https://ganjoor.net"))
|
||
{
|
||
string path = url.Replace("http://ganjoor.net", "").Replace("https://ganjoor.net", "");
|
||
int coupletNumber = -1;
|
||
string cleanPath = path;
|
||
|
||
// Extract couplet number if present
|
||
if (path.Contains("#bn"))
|
||
{
|
||
int coupletStartIndex = path.IndexOf("#bn") + "#bn".Length;
|
||
if (int.TryParse(path.Substring(coupletStartIndex), out coupletNumber))
|
||
{
|
||
cleanPath = path.Substring(0, path.IndexOf("#bn"));
|
||
}
|
||
}
|
||
|
||
// Remove trailing slash
|
||
if (cleanPath.Length > 0 && cleanPath[cleanPath.Length - 1] == '/')
|
||
cleanPath = cleanPath.Substring(0, cleanPath.Length - 1);
|
||
|
||
var page = await context.GanjoorPages
|
||
.AsNoTracking()
|
||
.Where(p => p.FullUrl == cleanPath)
|
||
.FirstOrDefaultAsync();
|
||
|
||
if (page != null)
|
||
{
|
||
string displayText = page.FullTitle;
|
||
|
||
// Add couplet summary if applicable
|
||
if (coupletNumber != -1)
|
||
{
|
||
string coupletSummary = await _GetCoupletSummary(page.Id, coupletNumber);
|
||
if (!string.IsNullOrEmpty(coupletSummary))
|
||
{
|
||
displayText = $"{page.FullTitle} » {coupletSummary}";
|
||
}
|
||
}
|
||
|
||
return $@"<a href=""{url}"" rel=""nofollow"">{displayText}</a>";
|
||
}
|
||
}
|
||
else
|
||
{
|
||
// External link - use generic text
|
||
return $@"<a href=""{url}"" rel=""nofollow noopener noreferrer"" target=""_blank"">پیوند به وبگاه بیرونی</a>";
|
||
}
|
||
}
|
||
|
||
// If link text was manually provided, keep it but still validate the link
|
||
return $@"<a href=""{url}"" rel=""nofollow noopener noreferrer"" target=""_blank"">{linkText}</a>";
|
||
}
|
||
|
||
private async Task<string> _GetCoupletSummary(int poemId, int coupletNumber)
|
||
{
|
||
int coupletIndex = coupletNumber - 1;
|
||
var verses = await _context.GanjoorVerses
|
||
.Where(v => v.PoemId == poemId)
|
||
.OrderBy(v => v.VOrder)
|
||
.ToListAsync();
|
||
|
||
int cIndex = -1;
|
||
for (int i = 0; i < verses.Count; i++)
|
||
{
|
||
if (verses[i].VersePosition != VersePosition.Left &&
|
||
verses[i].VersePosition != VersePosition.CenteredVerse2)
|
||
cIndex++;
|
||
|
||
if (cIndex == coupletIndex)
|
||
{
|
||
string coupletSummary = verses[i].Text;
|
||
|
||
if (verses[i].VersePosition == VersePosition.Right && i < verses.Count - 1)
|
||
{
|
||
coupletSummary += $" {verses[i + 1].Text}";
|
||
}
|
||
else if (verses[i].VersePosition == VersePosition.CenteredVerse1 &&
|
||
i < verses.Count - 1 &&
|
||
verses[i + 1].VersePosition == VersePosition.CenteredVerse2)
|
||
{
|
||
coupletSummary += $" {verses[i + 1].Text}";
|
||
}
|
||
|
||
return _CutSummary(coupletSummary);
|
||
}
|
||
}
|
||
|
||
return null;
|
||
}
|
||
|
||
private string _Linkify(string text)
|
||
{
|
||
// Improved Linkify with better URL detection
|
||
var urlRegex = new Regex(@"(https?://[^\s<>""']+)", RegexOptions.IgnoreCase);
|
||
return urlRegex.Replace(text, match =>
|
||
{
|
||
string url = match.Value;
|
||
return $@"<a href=""{url}"" rel=""nofollow noopener noreferrer"" target=""_blank"">{url}</a>";
|
||
});
|
||
}
|
||
|
||
/// <summary>
|
||
/// examine comments for long links
|
||
/// </summary>
|
||
/// <returns></returns>
|
||
public RServiceResult<bool> FindAndFixLongUrlsInComments()
|
||
{
|
||
_backgroundTaskQueue.QueueBackgroundWorkItem
|
||
(
|
||
async token =>
|
||
{
|
||
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>())) //this is long running job, so _context might be already been freed/collected by GC
|
||
{
|
||
LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context);
|
||
var job = (await jobProgressServiceEF.NewJob("FindAndFixLongUrlsInComments", "Query data")).Result;
|
||
|
||
try
|
||
{
|
||
var comments = //await context.GanjoorComments.Where(c => c.HtmlComment.Contains("href=")).ToArrayAsync();
|
||
await context.GanjoorComments.Where(c => c.HtmlComment.Contains("style")).ToArrayAsync();
|
||
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 0, $"Examining {comments.Length} Comments");
|
||
|
||
int percent = 0;
|
||
for (int i = 0; i < comments.Length; i++)
|
||
{
|
||
if (i * 100 / comments.Length > percent)
|
||
{
|
||
percent++;
|
||
await jobProgressServiceEF.UpdateJob(job.Id, percent);
|
||
}
|
||
|
||
var comment = comments[i];
|
||
|
||
var processedComment = await _ProcessCommentHtml(comment.HtmlComment, context);
|
||
|
||
if (processedComment.TextWasDropped)
|
||
{
|
||
// this is an unattended batch job re-sanitizing old comments,
|
||
// there is no user here to ask to fix their markup - leave the
|
||
// comment untouched rather than silently overwriting it with a
|
||
// mutilated version
|
||
continue;
|
||
}
|
||
|
||
if (processedComment.Html != comment.HtmlComment)
|
||
{
|
||
comment.HtmlComment = processedComment.Html;
|
||
context.Update(comment);
|
||
await context.SaveChangesAsync();
|
||
}
|
||
}
|
||
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true);
|
||
}
|
||
catch (Exception exp)
|
||
{
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString());
|
||
}
|
||
|
||
}
|
||
}
|
||
);
|
||
|
||
return new RServiceResult<bool>(true);
|
||
}
|
||
|
||
/// <summary>
|
||
/// start filling poems couplet indices
|
||
/// </summary>
|
||
/// <returns></returns>
|
||
public RServiceResult<bool> StartFillingPoemsCoupletIndices()
|
||
{
|
||
_backgroundTaskQueue.QueueBackgroundWorkItem
|
||
(
|
||
async token =>
|
||
{
|
||
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>())) //this is long running job, so _context might be already been freed/collected by GC
|
||
{
|
||
LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context);
|
||
var job = (await jobProgressServiceEF.NewJob("FillingPoemsCoupletIndices", "Query data")).Result;
|
||
|
||
try
|
||
{
|
||
var poemIds = await context.GanjoorPoems.AsNoTracking().Select(p => p.Id).ToListAsync();
|
||
|
||
|
||
int percent = 0;
|
||
for (int i = 0; i < poemIds.Count; i++)
|
||
{
|
||
if (i * 100 / poemIds.Count > percent)
|
||
{
|
||
percent++;
|
||
await jobProgressServiceEF.UpdateJob(job.Id, percent);
|
||
}
|
||
|
||
await _FillPoemCoupletIndices(context, poemIds[i]);
|
||
}
|
||
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true);
|
||
}
|
||
catch (Exception exp)
|
||
{
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString());
|
||
}
|
||
|
||
}
|
||
}
|
||
);
|
||
|
||
return new RServiceResult<bool>(true);
|
||
}
|
||
|
||
/// <summary>
|
||
/// refill couplet indices
|
||
/// </summary>
|
||
/// <param name="poemId"></param>
|
||
/// <returns></returns>
|
||
public async Task<RServiceResult<bool>> RefillCoupletIndicesAsync(int poemId)
|
||
{
|
||
try
|
||
{
|
||
await _FillPoemCoupletIndices(_context, poemId);
|
||
await _context.SaveChangesAsync();
|
||
return new RServiceResult<bool>(false);
|
||
}
|
||
catch (Exception exp)
|
||
{
|
||
return new RServiceResult<bool>(false, exp.ToString());
|
||
}
|
||
}
|
||
|
||
/// <summary>
|
||
/// fill section couplets count
|
||
/// </summary>
|
||
/// <returns></returns>
|
||
public RServiceResult<bool> StartFillingSectionCoupletCounts()
|
||
{
|
||
_backgroundTaskQueue.QueueBackgroundWorkItem
|
||
(
|
||
async token =>
|
||
{
|
||
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>())) //this is long running job, so _context might be already been freed/collected by GC
|
||
{
|
||
LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context);
|
||
var job = (await jobProgressServiceEF.NewJob("FillingSectionCoupletCounts", "Query data")).Result;
|
||
|
||
try
|
||
{
|
||
var sectionIds = await context.GanjoorPoemSections.AsNoTracking().Select(p => p.Id).ToListAsync();
|
||
|
||
|
||
int percent = 0;
|
||
for (int i = 0; i < sectionIds.Count; i++)
|
||
{
|
||
if (i * 100 / sectionIds.Count > percent)
|
||
{
|
||
percent++;
|
||
await jobProgressServiceEF.UpdateJob(job.Id, percent);
|
||
}
|
||
|
||
await _FillSectionCoupletCountsAsync(context, sectionIds[i]);
|
||
}
|
||
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true);
|
||
}
|
||
catch (Exception exp)
|
||
{
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString());
|
||
}
|
||
|
||
}
|
||
}
|
||
);
|
||
|
||
return new RServiceResult<bool>(true);
|
||
}
|
||
|
||
private async Task _FillSectionCoupletCountsAsync(RMuseumDbContext context, int sectionId)
|
||
{
|
||
var section = await context.GanjoorPoemSections.Where(s => s.Id == sectionId).SingleAsync();
|
||
|
||
section.CoupletsCount = await context.GanjoorVerses.Where(
|
||
v => v.PoemId == section.PoemId
|
||
&&
|
||
(
|
||
(section.VerseType == VersePoemSectionType.First && v.SectionIndex1 == section.Index)
|
||
||
|
||
(section.VerseType == VersePoemSectionType.Second && v.SectionIndex2 == section.Index)
|
||
||
|
||
(section.VerseType == VersePoemSectionType.Third && v.SectionIndex3 == section.Index)
|
||
||
|
||
(section.VerseType == VersePoemSectionType.Forth && v.SectionIndex4 == section.Index))
|
||
&&
|
||
(v.VersePosition == VersePosition.Left || v.VersePosition == VersePosition.CenteredVerse1)
|
||
).CountAsync();
|
||
|
||
context.Update(section);
|
||
}
|
||
|
||
|
||
/// <summary>
|
||
/// regenerate poem full titles to fix an old bug
|
||
/// </summary>
|
||
/// <returns></returns>
|
||
public RServiceResult<bool> RegeneratePoemsFullTitles()
|
||
{
|
||
_backgroundTaskQueue.QueueBackgroundWorkItem
|
||
(
|
||
async token =>
|
||
{
|
||
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>())) //this is long running job, so _context might be already been freed/collected by GC
|
||
{
|
||
LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context);
|
||
var job = (await jobProgressServiceEF.NewJob("RegenratePoemsFullTitles", "Query data")).Result;
|
||
int percent = 0;
|
||
try
|
||
{
|
||
var poemIds = await context.GanjoorPoems.AsNoTracking().Select(p => p.Id).ToListAsync();
|
||
|
||
for (int i = 0; i < poemIds.Count; i++)
|
||
{
|
||
if (i * 100 / poemIds.Count > percent)
|
||
{
|
||
percent++;
|
||
await jobProgressServiceEF.UpdateJob(job.Id, percent);
|
||
}
|
||
|
||
var poem = await context.GanjoorPoems.Where(p => p.Id == poemIds[i]).SingleOrDefaultAsync();
|
||
var catPage = await context.GanjoorPages.AsNoTracking().Where(p => p.GanjoorPageType == GanjoorPageType.CatPage && p.CatId == poem.CatId).SingleOrDefaultAsync();
|
||
if (catPage == null)
|
||
{
|
||
catPage = await context.GanjoorPages.AsNoTracking().Where(p => p.GanjoorPageType == GanjoorPageType.PoetPage && p.CatId == poem.CatId).SingleOrDefaultAsync();
|
||
}
|
||
var title = $"{catPage.FullTitle} » {poem.Title}";
|
||
if (title != poem.FullTitle)
|
||
{
|
||
poem.FullTitle = title;
|
||
context.Update(poem);
|
||
}
|
||
var page = await context.GanjoorPages.Where(p => p.Id == poem.Id).SingleOrDefaultAsync();
|
||
if (title != page.FullTitle)
|
||
{
|
||
page.FullTitle = title;
|
||
context.Update(page);
|
||
}
|
||
}
|
||
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true);
|
||
}
|
||
catch (Exception exp)
|
||
{
|
||
await jobProgressServiceEF.UpdateJob(job.Id, percent, "", false, exp.ToString());
|
||
}
|
||
|
||
}
|
||
}
|
||
);
|
||
|
||
return new RServiceResult<bool>(true);
|
||
}
|
||
|
||
|
||
/// <summary>
|
||
/// start finding rhymes for single couplets
|
||
/// </summary>
|
||
/// <returns></returns>
|
||
public RServiceResult<bool> StartFindingRhymesForSingleCouplets()
|
||
{
|
||
_backgroundTaskQueue.QueueBackgroundWorkItem
|
||
(
|
||
async token =>
|
||
{
|
||
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>())) //this is long running job, so _context might be already been freed/collected by GC
|
||
{
|
||
LongRunningJobProgressServiceEF jobProgressServiceEF = new LongRunningJobProgressServiceEF(context);
|
||
var job = (await jobProgressServiceEF.NewJob("FindingRhymesForSingleCouplets", "Query data")).Result;
|
||
|
||
try
|
||
{
|
||
var sections = await context.GanjoorPoemSections.AsNoTracking().Include(s => s.GanjoorMetre).Where(s => s.SectionType == PoemSectionType.WholePoem && s.GanjoorMetre != null && (s.RhymeLetters.Length >= 30 || s.RhymeLetters.Length < 2)).ToListAsync();
|
||
|
||
|
||
int percent = 0;
|
||
for (int i = 0; i < sections.Count; i++)
|
||
{
|
||
if (i * 100 / sections.Count > percent)
|
||
{
|
||
percent++;
|
||
await jobProgressServiceEF.UpdateJob(job.Id, percent);
|
||
}
|
||
|
||
var res = await _FindSectionRhyme(context, sections[i].Id);
|
||
if (string.IsNullOrEmpty(res.ExceptionString))
|
||
{
|
||
if (!string.IsNullOrEmpty(res.Result.Rhyme) && res.Result.Rhyme != sections[i].RhymeLetters)
|
||
{
|
||
var sectionTracked = await context.GanjoorPoemSections.Where(s => s.Id == sections[i].Id).SingleAsync();
|
||
var oldRhyme = sectionTracked.RhymeLetters;
|
||
sectionTracked.RhymeLetters = res.Result.Rhyme;
|
||
context.Update(sectionTracked);
|
||
await context.SaveChangesAsync();
|
||
await _UpdateRelatedSections(context, (int)sectionTracked.GanjoorMetreId, oldRhyme, jobProgressServiceEF, job, percent);
|
||
await _UpdateRelatedSections(context, (int)sectionTracked.GanjoorMetreId, res.Result.Rhyme, jobProgressServiceEF, job, percent);
|
||
}
|
||
}
|
||
}
|
||
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", true);
|
||
}
|
||
catch (Exception exp)
|
||
{
|
||
await jobProgressServiceEF.UpdateJob(job.Id, 100, "", false, exp.ToString());
|
||
}
|
||
|
||
}
|
||
}
|
||
);
|
||
|
||
return new RServiceResult<bool>(true);
|
||
}
|
||
}
|
||
} |