comment sanitzier

This commit is contained in:
Hamid Reza Mohammadi 2026-09-26 20:50:24 +03:30
parent 2ca4de21d6
commit 748b4b5b0a
5 changed files with 141 additions and 8 deletions

View File

@ -21096,6 +21096,31 @@
</summary>
<returns></returns>
</member>
<member name="M:RMuseum.Services.Implementation.GanjoorService._ExtractPlainText(System.String)">
<summary>
extracts normalized plain text from (possibly malformed) HTML the same way a
spec-compliant parser (and so also the sanitizer's own parser) sees it, so it can
be compared before/after sanitizing to detect whether sanitizing dropped real text
</summary>
</member>
<member name="M:RMuseum.Services.Implementation.GanjoorService._CommentSanitizationDroppedText(System.String,System.String)">
<summary>
true if sanitizing the comment dropped a meaningful chunk of the user's actual
text - not just markup/attributes, and not a harmless space lost when an inline
tag gets unwrapped. This happens when invalid/unclosed markup causes real comment
text to end up nested inside a tag the sanitizer correctly removes entirely
(e.g. a stray/unclosed tag that swallows the rest of the comment as its "content",
or an actually-disallowed tag such as script/style whose whole subtree is removed)
</summary>
</member>
<member name="M:RMuseum.Services.Implementation.GanjoorService._ProcessCommentHtml(System.String,RMuseum.DbContext.RMuseumDbContext)">
<summary>
sanitizes comment HTML; TextWasDropped is true when the sanitizing process removed
a meaningful chunk of the user's actual text (see _CommentSanitizationDroppedText) -
callers should not silently save Html in that case, but ask the user to fix their
markup instead
</summary>
</member>
<member name="M:RMuseum.Services.Implementation.GanjoorService.FindAndFixLongUrlsInComments">
<summary>
examine comments for long links

View File

@ -211,6 +211,20 @@ namespace RMuseum.Services.Implementation
{
return new RServiceResult<bool>(false, "bookmark not found");
}
// this note is typed in the same TinyMCE editor used for comments, so it needs
// the same HTML sanitizing (and the same guard against silently posting a
// half-baked note when sanitizing had to drop real text because of invalid markup)
if (!string.IsNullOrEmpty(note))
{
var processedNote = await _ProcessCommentHtml(note, _context);
if (processedNote.TextWasDropped)
{
return new RServiceResult<bool>(false, "بخشی از متن یادداشت شما به دلیل داشتن نشانه‌های HTML نامعتبر یا ناقص (مثلاً علامت‌های «کوچکتر از» یا «بزرگتر از» به‌تنهایی، یا برچسبی که بسته نشده) هنگام پاک‌سازی حذف شد. لطفاً متن را بررسی و اصلاح کنید و دوباره ثبت نمایید.");
}
note = processedNote.Html;
}
bookmark.PrivateNote = note;
_context.Update(bookmark);
await _context.SaveChangesAsync();

View File

@ -1,4 +1,5 @@
using Ganss.Xss;
using AngleSharp.Html.Parser;
using Ganss.Xss;
using Microsoft.EntityFrameworkCore;
using RMuseum.DbContext;
using RMuseum.Models.Ganjoor;
@ -402,8 +403,57 @@ namespace RMuseum.Services.Implementation
return new RServiceResult<bool>(true);
}
private async Task<string> _ProcessCommentHtml(string commentText, RMuseumDbContext context)
/// <summary>
/// extracts normalized plain text from (possibly malformed) HTML the same way a
/// spec-compliant parser (and so also the sanitizer's own parser) sees it, so it can
/// be compared before/after sanitizing to detect whether sanitizing dropped real text
/// </summary>
private static string _ExtractPlainText(string html)
{
if (string.IsNullOrWhiteSpace(html))
return "";
var parser = new HtmlParser();
using var document = parser.ParseDocument($"<body>{html}</body>");
string text = document.Body?.TextContent ?? "";
return Regex.Replace(text, @"\s+", " ").Trim();
}
/// <summary>
/// true if sanitizing the comment dropped a meaningful chunk of the user's actual
/// text - not just markup/attributes, and not a harmless space lost when an inline
/// tag gets unwrapped. This happens when invalid/unclosed markup causes real comment
/// text to end up nested inside a tag the sanitizer correctly removes entirely
/// (e.g. a stray/unclosed tag that swallows the rest of the comment as its "content",
/// or an actually-disallowed tag such as script/style whose whole subtree is removed)
/// </summary>
private static bool _CommentSanitizationDroppedText(string originalPlainText, string sanitizedPlainText)
{
originalPlainText = (originalPlainText ?? "").Trim();
sanitizedPlainText = (sanitizedPlainText ?? "").Trim();
if (originalPlainText.Length == 0)
return false; // nothing to lose
if (sanitizedPlainText.Length == 0)
return true; // the whole comment text vanished
if (originalPlainText.Length - sanitizedPlainText.Length <= 2)
return false; // negligible, e.g. a boundary space lost when unwrapping a tag
return sanitizedPlainText.Length < originalPlainText.Length * 0.95;
}
/// <summary>
/// sanitizes comment HTML; TextWasDropped is true when the sanitizing process removed
/// a meaningful chunk of the user's actual text (see _CommentSanitizationDroppedText) -
/// callers should not silently save Html in that case, but ask the user to fix their
/// markup instead
/// </summary>
private async Task<(string Html, bool TextWasDropped)> _ProcessCommentHtml(string commentText, RMuseumDbContext context)
{
string originalPlainText = _ExtractPlainText(commentText);
// Use a proper HTML sanitizer
var sanitizer = new HtmlSanitizer();
@ -442,10 +492,16 @@ namespace RMuseum.Services.Implementation
// Sanitize the HTML
string sanitizedHtml = sanitizer.Sanitize(commentText);
// Detect whether sanitizing took real text down along with the invalid markup it
// removed, before Linkify/internal-link processing below can itself change the
// visible text (e.g. replacing a bare URL's text with a page title) in a way that
// is not a loss.
bool textWasDropped = _CommentSanitizationDroppedText(originalPlainText, _ExtractPlainText(sanitizedHtml));
// Process URLs (Linkify) and internal Ganjoor links
sanitizedHtml = await _ProcessUrls(sanitizedHtml, context);
return sanitizedHtml;
return (sanitizedHtml, textWasDropped);
}
private async Task<string> _ProcessUrls(string html, RMuseumDbContext context)
@ -630,11 +686,20 @@ namespace RMuseum.Services.Implementation
var comment = comments[i];
string commentText = await _ProcessCommentHtml(comment.HtmlComment, context);
var processedComment = await _ProcessCommentHtml(comment.HtmlComment, context);
if (commentText != comment.HtmlComment)
if (processedComment.TextWasDropped)
{
comment.HtmlComment = commentText;
// this is an unattended batch job re-sanitizing old comments,
// there is no user here to ask to fix their markup - leave the
// comment untouched rather than silently overwriting it with a
// mutilated version
continue;
}
if (processedComment.Html != comment.HtmlComment)
{
comment.HtmlComment = processedComment.Html;
context.Update(comment);
await context.SaveChangesAsync();
}

View File

@ -160,6 +160,17 @@ namespace RMuseum.Services.Implementation
{
try
{
// this is typed in the same TinyMCE editor used for comments, and once
// published it is shown to anonymous visitors on the poet's page, so it needs
// the same sanitizing (and the same guard against silently keeping a half-baked
// suggestion when sanitizing had to drop real text because of invalid markup)
var processedContents = await _ProcessCommentHtml(model.Contents, _context);
if (processedContents.TextWasDropped)
{
return new RServiceResult<GanjoorPoetSuggestedSpecLineViewModel>(null, "بخشی از متن پیشنهادی شما به دلیل داشتن نشانه‌های HTML نامعتبر یا ناقص (مثلاً علامت‌های «کوچکتر از» یا «بزرگتر از» به‌تنهایی، یا برچسبی که بسته نشده) هنگام پاک‌سازی حذف شد. لطفاً متن را بررسی و اصلاح کنید و دوباره ارسال نمایید.");
}
model.Contents = processedContents.Html;
var dbModel = new GanjoorPoetSuggestedSpecLine()
{
PoetId = model.PoetId,
@ -208,6 +219,14 @@ namespace RMuseum.Services.Implementation
{
var dbModel = await _context.GanjoorPoetSuggestedSpecLines.Where(s => s.Id == model.Id).SingleAsync();
var processedContents = await _ProcessCommentHtml(model.Contents, _context);
if (processedContents.TextWasDropped)
{
return new RServiceResult<bool>(false, "بخشی از متن به دلیل داشتن نشانه‌های HTML نامعتبر یا ناقص (مثلاً علامت‌های «کوچکتر از» یا «بزرگتر از» به‌تنهایی، یا برچسبی که بسته نشده) هنگام پاک‌سازی حذف شد. لطفاً متن را بررسی و اصلاح کنید و دوباره ثبت نمایید.");
}
model.Contents = processedContents.Html;
bool publishIsChanged = model.Published != dbModel.Published;
if (publishIsChanged)
{

View File

@ -1041,7 +1041,12 @@ namespace RMuseum.Services.Implementation
content = content.ApplyCorrectYeKe();
content = await _ProcessCommentHtml(content, _context);
var processedComment = await _ProcessCommentHtml(content, _context);
if (processedComment.TextWasDropped)
{
return new RServiceResult<GanjoorCommentSummaryViewModel>(null, "بخشی از متن حاشیهٔ شما به دلیل داشتن نشانه‌های HTML نامعتبر یا ناقص (مثلاً علامت‌های «کوچکتر از» یا «بزرگتر از» به‌تنهایی، یا برچسبی که بسته نشده) هنگام پاک‌سازی حذف شد. لطفاً متن حاشیه را بررسی و اصلاح کنید و دوباره ارسال نمایید.");
}
content = processedComment.Html;
string commentText = System.Net.WebUtility.HtmlDecode(Regex.Replace(content, "<.*?>", string.Empty));
@ -1181,7 +1186,12 @@ namespace RMuseum.Services.Implementation
htmlComment = htmlComment.ApplyCorrectYeKe();
htmlComment = await _ProcessCommentHtml(htmlComment, _context);
var processedComment = await _ProcessCommentHtml(htmlComment, _context);
if (processedComment.TextWasDropped)
{
return new RServiceResult<bool>(false, "بخشی از متن حاشیهٔ شما به دلیل داشتن نشانه‌های HTML نامعتبر یا ناقص (مثلاً علامت‌های «کوچکتر از» یا «بزرگتر از» به‌تنهایی، یا برچسبی که بسته نشده) هنگام پاک‌سازی حذف شد. لطفاً متن حاشیه را بررسی و اصلاح کنید و دوباره ارسال نمایید.");
}
htmlComment = processedComment.Html;
string commentText = System.Net.WebUtility.HtmlDecode(Regex.Replace(htmlComment, "<.*?>", string.Empty));