adding html sanitizer to comments

This commit is contained in:
Hamid Reza Mohammadi 2026-07-20 18:35:59 +03:30
parent d7bb49b771
commit 0d2f7b7ff9
2 changed files with 202 additions and 225 deletions

View File

@ -21,6 +21,7 @@
<PackageReference Include="Betalgo.Ranul.OpenAI" Version="9.2.6" /> <PackageReference Include="Betalgo.Ranul.OpenAI" Version="9.2.6" />
<PackageReference Include="DNTPersianUtils.Core" Version="7.0.0" /> <PackageReference Include="DNTPersianUtils.Core" Version="7.0.0" />
<PackageReference Include="FluentFTP" Version="54.2.0" /> <PackageReference Include="FluentFTP" Version="54.2.0" />
<PackageReference Include="HtmlSanitizer" Version="9.0.892" />
<PackageReference Include="Microsoft.Data.Sqlite" Version="10.0.9" /> <PackageReference Include="Microsoft.Data.Sqlite" Version="10.0.9" />
<PackageReference Include="Microsoft.EntityFrameworkCore.Design" Version="10.0.9"> <PackageReference Include="Microsoft.EntityFrameworkCore.Design" Version="10.0.9">
<PrivateAssets>all</PrivateAssets> <PrivateAssets>all</PrivateAssets>

View File

@ -1,4 +1,5 @@
using Microsoft.EntityFrameworkCore; using Ganss.Xss;
using Microsoft.EntityFrameworkCore;
using RMuseum.DbContext; using RMuseum.DbContext;
using RMuseum.Models.Ganjoor; using RMuseum.Models.Ganjoor;
using RSecurityBackend.Models.Generic; using RSecurityBackend.Models.Generic;
@ -9,6 +10,7 @@ using System.Collections.Generic;
using System.Data; using System.Data;
using System.Linq; using System.Linq;
using System.Net.Http; using System.Net.Http;
using System.Text.RegularExpressions;
using System.Threading.Tasks; using System.Threading.Tasks;
namespace RMuseum.Services.Implementation namespace RMuseum.Services.Implementation
@ -370,223 +372,197 @@ namespace RMuseum.Services.Implementation
private async Task<string> _ProcessCommentHtml(string commentText, RMuseumDbContext context) private async Task<string> _ProcessCommentHtml(string commentText, RMuseumDbContext context)
{ {
string[] allowedTags = new string[] // Use a proper HTML sanitizer
{ var sanitizer = new HtmlSanitizer();
"p",
"a",
"br",
"b",
"i",
"strong",
"img"
};
if (commentText.IndexOf("<") != -1)
{
int openTagIndex = commentText.IndexOf('<');
while (openTagIndex != -1)
{
int closeOpenningTagIndex = commentText.IndexOf('>', openTagIndex + 1);
if (closeOpenningTagIndex == -1) //an unclosed tag
{
if (commentText.IndexOf(' ', openTagIndex + 1) != -1)
{
commentText = commentText.Substring(0, openTagIndex) + commentText.Substring(commentText.IndexOf(' ', openTagIndex + 1));
}
else
{
commentText = commentText.Substring(0, openTagIndex);
}
}
else
{
int anotherOpenTagInBetweenIndex = commentText.IndexOf('<', openTagIndex + 1);
if (anotherOpenTagInBetweenIndex != -1 && anotherOpenTagInBetweenIndex < closeOpenningTagIndex)
{
commentText = commentText.Substring(0, openTagIndex) + commentText.Substring(anotherOpenTagInBetweenIndex);
}
else
{
int tagTypeCloseIndex = closeOpenningTagIndex;
int spaceAfterOpenningTagIndex = commentText.IndexOf(' ', openTagIndex + 1);
if (spaceAfterOpenningTagIndex != -1 && spaceAfterOpenningTagIndex < tagTypeCloseIndex)
tagTypeCloseIndex = spaceAfterOpenningTagIndex;
// Configure allowed tags
sanitizer.AllowedTags.Clear();
sanitizer.AllowedTags.Add("p");
sanitizer.AllowedTags.Add("a");
sanitizer.AllowedTags.Add("br");
sanitizer.AllowedTags.Add("b");
sanitizer.AllowedTags.Add("i");
sanitizer.AllowedTags.Add("strong");
sanitizer.AllowedTags.Add("img");
sanitizer.AllowedTags.Add("span");
string tagType = commentText.Substring(openTagIndex + 1, tagTypeCloseIndex - openTagIndex - 1).ToLower(); // Configure allowed attributes
tagType = tagType.Replace("/", "");//include close tags sanitizer.AllowedAttributes.Clear();
if (tagType.Length == 0) sanitizer.AllowedAttributes.Add("href");
{ sanitizer.AllowedAttributes.Add("src");
if (closeOpenningTagIndex == commentText.Length - 1) sanitizer.AllowedAttributes.Add("alt");
commentText = commentText.Substring(0, openTagIndex); sanitizer.AllowedAttributes.Add("title");
else sanitizer.AllowedAttributes.Add("rel");
commentText = commentText.Substring(0, openTagIndex) + commentText.Substring(closeOpenningTagIndex + 1);
} // IMPORTANT: Remove all style attributes to prevent colored text and font changes
else sanitizer.AllowedAttributes.Remove("style");
{
if (!allowedTags.Contains(tagType)) // Disallow any CSS or style-related attributes
{ sanitizer.AllowedCssProperties.Clear();
commentText = commentText.Substring(0, openTagIndex) + commentText.Substring(closeOpenningTagIndex + 1);
commentText = commentText.Replace($"</{tagType}>", ""); // Additional security: disallow data URIs and javascript: protocols
} sanitizer.AllowedSchemes.Clear();
} sanitizer.AllowedSchemes.Add("http");
} sanitizer.AllowedSchemes.Add("https");
sanitizer.AllowedSchemes.Add("mailto");
sanitizer.AllowedSchemes.Add("ftp");
// Sanitize the HTML
string sanitizedHtml = sanitizer.Sanitize(commentText);
// Process URLs (Linkify) and internal Ganjoor links
sanitizedHtml = await _ProcessUrls(sanitizedHtml, context);
return sanitizedHtml;
} }
private async Task<string> _ProcessUrls(string html, RMuseumDbContext context)
{
// First, process any existing href links
html = await _ProcessExistingLinks(html, context);
openTagIndex = commentText.IndexOf("<", openTagIndex + 1); // Then linkify any bare URLs that aren't already in links
} if (!html.Contains("href=\"") && html.Contains("http"))
{
html = _Linkify(html);
} }
if (commentText.IndexOf("href=") == -1 && commentText.IndexOf("http") != -1) return html;
{
commentText = _Linkify(commentText);
} }
int index = commentText.IndexOf("href=");
while (index != -1) private async Task<string> _ProcessExistingLinks(string html, RMuseumDbContext context)
{ {
index += "href=\"".Length; // Use regex to find and process all href attributes
commentText = commentText.Replace("'", "\""); var hrefRegex = new Regex(@"<a\s+(?:[^>]*?\s+)?href=""([^""]*)""[^>]*>(.*?)</a>", RegexOptions.IgnoreCase | RegexOptions.Singleline);
if (commentText.IndexOf("\"", index) != -1)
var matches = hrefRegex.Matches(html);
var result = html;
var replacements = new List<(string old, string @new)>();
foreach (Match match in matches)
{ {
int closeIndex = commentText.IndexOf("\"", index); string url = match.Groups[1].Value;
if (closeIndex == -1) string linkText = match.Groups[2].Value;
{ string newLink = await _ProcessSingleLink(url, linkText, context);
continue;
// Store the replacement
replacements.Add((match.Value, newLink));
} }
string url = commentText.Substring(index, closeIndex - index);
closeIndex = commentText.IndexOf(">", index); // Apply replacements from end to start to maintain indices
if (closeIndex != -1 && commentText.IndexOf("</a>", closeIndex) != -1) foreach (var replacement in replacements.OrderByDescending(r => result.IndexOf(r.old)))
{ {
closeIndex += ">".Length; result = result.Replace(replacement.old, replacement.@new);
string urlText = commentText.Substring(closeIndex, commentText.IndexOf("</a>", closeIndex) - closeIndex); }
if (urlText == url)
return result;
}
private async Task<string> _ProcessSingleLink(string url, string linkText, RMuseumDbContext context)
{ {
bool textFixed = false; // If link text is the same as URL (auto-generated link), try to improve it
if (urlText.IndexOf("http://ganjoor.net") == 0 || urlText.IndexOf("https://ganjoor.net") == 0) if (url == linkText || linkText.Trim() == url.Trim())
{ {
urlText = urlText.Replace("http://ganjoor.net", "").Replace("https://ganjoor.net", ""); // Process Ganjoor internal links
if (url.StartsWith("http://ganjoor.net") || url.StartsWith("https://ganjoor.net"))
{
string path = url.Replace("http://ganjoor.net", "").Replace("https://ganjoor.net", "");
int coupletNumber = -1; int coupletNumber = -1;
if (urlText.IndexOf("#bn") != -1) string cleanPath = path;
// Extract couplet number if present
if (path.Contains("#bn"))
{ {
int coupletStartIndex = urlText.IndexOf("#bn") + "#bn".Length; int coupletStartIndex = path.IndexOf("#bn") + "#bn".Length;
if (int.TryParse(urlText.Substring(coupletStartIndex), out coupletNumber)) if (int.TryParse(path.Substring(coupletStartIndex), out coupletNumber))
{ {
urlText = urlText.Substring(0, urlText.IndexOf("#bn")); cleanPath = path.Substring(0, path.IndexOf("#bn"));
} }
} }
if (urlText.Length > 0 && urlText[urlText.Length - 1] == '/')
urlText = urlText.Substring(0, urlText.Length - 1); // Remove trailing slash
var page = await context.GanjoorPages.AsNoTracking().Where(p => p.FullUrl == urlText).FirstOrDefaultAsync(); if (cleanPath.Length > 0 && cleanPath[cleanPath.Length - 1] == '/')
cleanPath = cleanPath.Substring(0, cleanPath.Length - 1);
var page = await context.GanjoorPages
.AsNoTracking()
.Where(p => p.FullUrl == cleanPath)
.FirstOrDefaultAsync();
if (page != null) if (page != null)
{ {
string displayText = page.FullTitle;
// Add couplet summary if applicable
if (coupletNumber != -1) if (coupletNumber != -1)
{ {
string coupletSummary = ""; string coupletSummary = await _GetCoupletSummary(page.Id, coupletNumber);
if (!string.IsNullOrEmpty(coupletSummary))
{
displayText = $"{page.FullTitle} » {coupletSummary}";
}
}
return $@"<a href=""{url}"" rel=""nofollow"">{displayText}</a>";
}
}
else
{
// External link - use generic text
return $@"<a href=""{url}"" rel=""nofollow noopener noreferrer"" target=""_blank"">پیوند به وبگاه بیرونی</a>";
}
}
// If link text was manually provided, keep it but still validate the link
return $@"<a href=""{url}"" rel=""nofollow noopener noreferrer"" target=""_blank"">{linkText}</a>";
}
private async Task<string> _GetCoupletSummary(int poemId, int coupletNumber)
{
int coupletIndex = coupletNumber - 1; int coupletIndex = coupletNumber - 1;
var verses = await _context.GanjoorVerses.Where(v => v.PoemId == page.Id).OrderBy(v => v.VOrder).ToListAsync(); var verses = await _context.GanjoorVerses
.Where(v => v.PoemId == poemId)
.OrderBy(v => v.VOrder)
.ToListAsync();
int cIndex = -1; int cIndex = -1;
for (int i = 0; i < verses.Count; i++) for (int i = 0; i < verses.Count; i++)
{ {
if (verses[i].VersePosition != VersePosition.Left && verses[i].VersePosition != VersePosition.CenteredVerse2) if (verses[i].VersePosition != VersePosition.Left &&
verses[i].VersePosition != VersePosition.CenteredVerse2)
cIndex++; cIndex++;
if (cIndex == coupletIndex) if (cIndex == coupletIndex)
{ {
coupletSummary = verses[i].Text; string coupletSummary = verses[i].Text;
if (verses[i].VersePosition == VersePosition.Right)
{ if (verses[i].VersePosition == VersePosition.Right && i < verses.Count - 1)
if (i < verses.Count - 1)
{ {
coupletSummary += $" {verses[i + 1].Text}"; coupletSummary += $" {verses[i + 1].Text}";
} }
} else if (verses[i].VersePosition == VersePosition.CenteredVerse1 &&
if (verses[i].VersePosition == VersePosition.CenteredVerse1) i < verses.Count - 1 &&
{ verses[i + 1].VersePosition == VersePosition.CenteredVerse2)
if (i < verses.Count - 1)
{
if (verses[i + 1].VersePosition == VersePosition.CenteredVerse2)
{ {
coupletSummary += $" {verses[i + 1].Text}"; coupletSummary += $" {verses[i + 1].Text}";
} }
}
} return _CutSummary(coupletSummary);
break;
}
}
if (!string.IsNullOrEmpty(coupletSummary))
{
coupletSummary = _CutSummary(coupletSummary);
commentText = commentText.Substring(0, closeIndex) + page.FullTitle + " » " + coupletSummary + commentText.Substring(commentText.IndexOf("</a>", closeIndex));
textFixed = true;
}
else
coupletNumber = -1;
}
if (coupletNumber == -1)
{
commentText = commentText.Substring(0, closeIndex) + page.FullTitle + commentText.Substring(commentText.IndexOf("</a>", closeIndex));
textFixed = true;
} }
} }
} return null;
if (!textFixed)
commentText = commentText.Substring(0, closeIndex) + "پیوند به وبگاه بیرونی" + commentText.Substring(commentText.IndexOf("</a>", closeIndex));
}
}
}
index = commentText.IndexOf("href=\"", index);
}
return commentText;
} }
private string _Linkify(string SearchText) private string _Linkify(string text)
{ {
if (SearchText.IndexOf("href") != -1) // Improved Linkify with better URL detection
return SearchText; var urlRegex = new Regex(@"(https?://[^\s<>""']+)", RegexOptions.IgnoreCase);
int linkIndex = SearchText.IndexOf("http"); return urlRegex.Replace(text, match =>
while (linkIndex != -1)
{ {
int linkEndIndex = SearchText.IndexOfAny(new char[] { '\r', '\n', '<', ' ' }, linkIndex); string url = match.Value;
if (linkEndIndex == -1) return $@"<a href=""{url}"" rel=""nofollow noopener noreferrer"" target=""_blank"">{url}</a>";
linkEndIndex = SearchText.Length - 1; });
if (linkEndIndex != -1)
{
string link = SearchText.Substring(linkIndex, linkEndIndex - linkIndex);
SearchText
=
SearchText.Substring(0, linkIndex)
+
"<a href=\""
+
link
+
"\" rel=\"nofollow\">"
+
link
+
"</a>"
+
SearchText.Substring(linkEndIndex);
linkIndex =
(
SearchText.Substring(0, linkIndex)
+
"<a href=\""
+
link
+
"\" rel=\"nofollow\">"
+
link
+
"</a>"
).Length;
linkIndex = SearchText.IndexOf("http", linkIndex);
}
else
linkIndex = SearchText.IndexOf("http", linkIndex + "http".Length);
}
return SearchText;
} }
/// <summary> /// <summary>