using System.Collections.Generic; using System.Linq; using System.Text.RegularExpressions; using System.Web; namespace RMuseum.Utils { public class DivanHtmlTools { public static string StripHtmlTags(string input) { // Remove HTML tags using Regex string textWithoutTags = Regex.Replace(input, "<.*?>", string.Empty); // Decode HTML entities (e.g., & → &) return HttpUtility.HtmlDecode(textWithoutTags); } public static List ExtractLinksWithRegex(string html) { List links = new List(); string pattern = @"<(a|img)\b[^>]*?\b(href|src)\s*=\s*[""']?([^""' >]+)[""']?"; MatchCollection matches = Regex.Matches(html, pattern, RegexOptions.IgnoreCase); foreach (Match match in matches) { if (match.Groups.Count >= 4) { string url = match.Groups[3].Value; if (!string.IsNullOrEmpty(url)) links.Add(url); } } return links.Distinct().ToList(); // Remove duplicates } } }