diff --git a/RMuseum/RMuseum.xml b/RMuseum/RMuseum.xml
index 167762c1..f9ae3ca3 100644
--- a/RMuseum/RMuseum.xml
+++ b/RMuseum/RMuseum.xml
@@ -21096,6 +21096,31 @@
+
+
+ extracts normalized plain text from (possibly malformed) HTML the same way a
+ spec-compliant parser (and so also the sanitizer's own parser) sees it, so it can
+ be compared before/after sanitizing to detect whether sanitizing dropped real text
+
+
+
+
+ true if sanitizing the comment dropped a meaningful chunk of the user's actual
+ text - not just markup/attributes, and not a harmless space lost when an inline
+ tag gets unwrapped. This happens when invalid/unclosed markup causes real comment
+ text to end up nested inside a tag the sanitizer correctly removes entirely
+ (e.g. a stray/unclosed tag that swallows the rest of the comment as its "content",
+ or an actually-disallowed tag such as script/style whose whole subtree is removed)
+
+
+
+
+ sanitizes comment HTML; TextWasDropped is true when the sanitizing process removed
+ a meaningful chunk of the user's actual text (see _CommentSanitizationDroppedText) -
+ callers should not silently save Html in that case, but ask the user to fix their
+ markup instead
+
+
examine comments for long links
diff --git a/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-Bookmarking.cs b/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-Bookmarking.cs
index 00f41427..42441dab 100644
--- a/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-Bookmarking.cs
+++ b/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-Bookmarking.cs
@@ -211,6 +211,20 @@ namespace RMuseum.Services.Implementation
{
return new RServiceResult(false, "bookmark not found");
}
+
+ // this note is typed in the same TinyMCE editor used for comments, so it needs
+ // the same HTML sanitizing (and the same guard against silently posting a
+ // half-baked note when sanitizing had to drop real text because of invalid markup)
+ if (!string.IsNullOrEmpty(note))
+ {
+ var processedNote = await _ProcessCommentHtml(note, _context);
+ if (processedNote.TextWasDropped)
+ {
+ return new RServiceResult(false, "بخشی از متن یادداشت شما به دلیل داشتن نشانههای HTML نامعتبر یا ناقص (مثلاً علامتهای «کوچکتر از» یا «بزرگتر از» بهتنهایی، یا برچسبی که بسته نشده) هنگام پاکسازی حذف شد. لطفاً متن را بررسی و اصلاح کنید و دوباره ثبت نمایید.");
+ }
+ note = processedNote.Html;
+ }
+
bookmark.PrivateNote = note;
_context.Update(bookmark);
await _context.SaveChangesAsync();
diff --git a/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-Maintenance.cs b/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-Maintenance.cs
index a86f04aa..655f0dac 100644
--- a/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-Maintenance.cs
+++ b/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-Maintenance.cs
@@ -1,4 +1,5 @@
-using Ganss.Xss;
+using AngleSharp.Html.Parser;
+using Ganss.Xss;
using Microsoft.EntityFrameworkCore;
using RMuseum.DbContext;
using RMuseum.Models.Ganjoor;
@@ -402,8 +403,57 @@ namespace RMuseum.Services.Implementation
return new RServiceResult(true);
}
- private async Task _ProcessCommentHtml(string commentText, RMuseumDbContext context)
+ ///
+ /// extracts normalized plain text from (possibly malformed) HTML the same way a
+ /// spec-compliant parser (and so also the sanitizer's own parser) sees it, so it can
+ /// be compared before/after sanitizing to detect whether sanitizing dropped real text
+ ///
+ private static string _ExtractPlainText(string html)
{
+ if (string.IsNullOrWhiteSpace(html))
+ return "";
+
+ var parser = new HtmlParser();
+ using var document = parser.ParseDocument($"{html}");
+ string text = document.Body?.TextContent ?? "";
+ return Regex.Replace(text, @"\s+", " ").Trim();
+ }
+
+ ///
+ /// true if sanitizing the comment dropped a meaningful chunk of the user's actual
+ /// text - not just markup/attributes, and not a harmless space lost when an inline
+ /// tag gets unwrapped. This happens when invalid/unclosed markup causes real comment
+ /// text to end up nested inside a tag the sanitizer correctly removes entirely
+ /// (e.g. a stray/unclosed tag that swallows the rest of the comment as its "content",
+ /// or an actually-disallowed tag such as script/style whose whole subtree is removed)
+ ///
+ private static bool _CommentSanitizationDroppedText(string originalPlainText, string sanitizedPlainText)
+ {
+ originalPlainText = (originalPlainText ?? "").Trim();
+ sanitizedPlainText = (sanitizedPlainText ?? "").Trim();
+
+ if (originalPlainText.Length == 0)
+ return false; // nothing to lose
+
+ if (sanitizedPlainText.Length == 0)
+ return true; // the whole comment text vanished
+
+ if (originalPlainText.Length - sanitizedPlainText.Length <= 2)
+ return false; // negligible, e.g. a boundary space lost when unwrapping a tag
+
+ return sanitizedPlainText.Length < originalPlainText.Length * 0.95;
+ }
+
+ ///
+ /// sanitizes comment HTML; TextWasDropped is true when the sanitizing process removed
+ /// a meaningful chunk of the user's actual text (see _CommentSanitizationDroppedText) -
+ /// callers should not silently save Html in that case, but ask the user to fix their
+ /// markup instead
+ ///
+ private async Task<(string Html, bool TextWasDropped)> _ProcessCommentHtml(string commentText, RMuseumDbContext context)
+ {
+ string originalPlainText = _ExtractPlainText(commentText);
+
// Use a proper HTML sanitizer
var sanitizer = new HtmlSanitizer();
@@ -442,10 +492,16 @@ namespace RMuseum.Services.Implementation
// Sanitize the HTML
string sanitizedHtml = sanitizer.Sanitize(commentText);
+ // Detect whether sanitizing took real text down along with the invalid markup it
+ // removed, before Linkify/internal-link processing below can itself change the
+ // visible text (e.g. replacing a bare URL's text with a page title) in a way that
+ // is not a loss.
+ bool textWasDropped = _CommentSanitizationDroppedText(originalPlainText, _ExtractPlainText(sanitizedHtml));
+
// Process URLs (Linkify) and internal Ganjoor links
sanitizedHtml = await _ProcessUrls(sanitizedHtml, context);
- return sanitizedHtml;
+ return (sanitizedHtml, textWasDropped);
}
private async Task _ProcessUrls(string html, RMuseumDbContext context)
@@ -630,11 +686,20 @@ namespace RMuseum.Services.Implementation
var comment = comments[i];
- string commentText = await _ProcessCommentHtml(comment.HtmlComment, context);
+ var processedComment = await _ProcessCommentHtml(comment.HtmlComment, context);
- if (commentText != comment.HtmlComment)
+ if (processedComment.TextWasDropped)
{
- comment.HtmlComment = commentText;
+ // this is an unattended batch job re-sanitizing old comments,
+ // there is no user here to ask to fix their markup - leave the
+ // comment untouched rather than silently overwriting it with a
+ // mutilated version
+ continue;
+ }
+
+ if (processedComment.Html != comment.HtmlComment)
+ {
+ comment.HtmlComment = processedComment.Html;
context.Update(comment);
await context.SaveChangesAsync();
}
diff --git a/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-PhotosProject.cs b/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-PhotosProject.cs
index 60375a1b..7ab8e24a 100644
--- a/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-PhotosProject.cs
+++ b/RMuseum/Services/Implementation/GanjoorService-Partials/GanjoorService-PhotosProject.cs
@@ -160,6 +160,17 @@ namespace RMuseum.Services.Implementation
{
try
{
+ // this is typed in the same TinyMCE editor used for comments, and once
+ // published it is shown to anonymous visitors on the poet's page, so it needs
+ // the same sanitizing (and the same guard against silently keeping a half-baked
+ // suggestion when sanitizing had to drop real text because of invalid markup)
+ var processedContents = await _ProcessCommentHtml(model.Contents, _context);
+ if (processedContents.TextWasDropped)
+ {
+ return new RServiceResult(null, "بخشی از متن پیشنهادی شما به دلیل داشتن نشانههای HTML نامعتبر یا ناقص (مثلاً علامتهای «کوچکتر از» یا «بزرگتر از» بهتنهایی، یا برچسبی که بسته نشده) هنگام پاکسازی حذف شد. لطفاً متن را بررسی و اصلاح کنید و دوباره ارسال نمایید.");
+ }
+ model.Contents = processedContents.Html;
+
var dbModel = new GanjoorPoetSuggestedSpecLine()
{
PoetId = model.PoetId,
@@ -208,6 +219,14 @@ namespace RMuseum.Services.Implementation
{
var dbModel = await _context.GanjoorPoetSuggestedSpecLines.Where(s => s.Id == model.Id).SingleAsync();
+
+ var processedContents = await _ProcessCommentHtml(model.Contents, _context);
+ if (processedContents.TextWasDropped)
+ {
+ return new RServiceResult(false, "بخشی از متن به دلیل داشتن نشانههای HTML نامعتبر یا ناقص (مثلاً علامتهای «کوچکتر از» یا «بزرگتر از» بهتنهایی، یا برچسبی که بسته نشده) هنگام پاکسازی حذف شد. لطفاً متن را بررسی و اصلاح کنید و دوباره ثبت نمایید.");
+ }
+ model.Contents = processedContents.Html;
+
bool publishIsChanged = model.Published != dbModel.Published;
if (publishIsChanged)
{
diff --git a/RMuseum/Services/Implementation/GanjoorService.cs b/RMuseum/Services/Implementation/GanjoorService.cs
index 6791df8d..afd576c8 100644
--- a/RMuseum/Services/Implementation/GanjoorService.cs
+++ b/RMuseum/Services/Implementation/GanjoorService.cs
@@ -1041,7 +1041,12 @@ namespace RMuseum.Services.Implementation
content = content.ApplyCorrectYeKe();
- content = await _ProcessCommentHtml(content, _context);
+ var processedComment = await _ProcessCommentHtml(content, _context);
+ if (processedComment.TextWasDropped)
+ {
+ return new RServiceResult(null, "بخشی از متن حاشیهٔ شما به دلیل داشتن نشانههای HTML نامعتبر یا ناقص (مثلاً علامتهای «کوچکتر از» یا «بزرگتر از» بهتنهایی، یا برچسبی که بسته نشده) هنگام پاکسازی حذف شد. لطفاً متن حاشیه را بررسی و اصلاح کنید و دوباره ارسال نمایید.");
+ }
+ content = processedComment.Html;
string commentText = System.Net.WebUtility.HtmlDecode(Regex.Replace(content, "<.*?>", string.Empty));
@@ -1181,7 +1186,12 @@ namespace RMuseum.Services.Implementation
htmlComment = htmlComment.ApplyCorrectYeKe();
- htmlComment = await _ProcessCommentHtml(htmlComment, _context);
+ var processedComment = await _ProcessCommentHtml(htmlComment, _context);
+ if (processedComment.TextWasDropped)
+ {
+ return new RServiceResult(false, "بخشی از متن حاشیهٔ شما به دلیل داشتن نشانههای HTML نامعتبر یا ناقص (مثلاً علامتهای «کوچکتر از» یا «بزرگتر از» بهتنهایی، یا برچسبی که بسته نشده) هنگام پاکسازی حذف شد. لطفاً متن حاشیه را بررسی و اصلاح کنید و دوباره ارسال نمایید.");
+ }
+ htmlComment = processedComment.Html;
string commentText = System.Net.WebUtility.HtmlDecode(Regex.Replace(htmlComment, "<.*?>", string.Empty));