Remote HTML is sanitized before it is stored

S5 of the roadmap. ContentSanitizer wraps HtmlSanitizer with Mastodon's
allowlist: the inline and list tags Mastodon keeps, href/rel/class and
the list attributes, microformat and mention/hashtag/ellipsis/invisible
classes, Mastodon's link schemes, every link rel=nofollow noopener
noreferrer, relative links unlinked, headings folded to a bold paragraph,
and the contents of script, style, svg, iframe and friends dropped rather
than kept as text.

Post and DmPost gain ContentHtml (what is shown) and ContentFormat
(Markdown for local, Html for remote). Inbound Create and Update, and a
remote actor's biography, are sanitized on the way in; local posts store
their Markdig rendering. Migration _002 does the same to what is already
stored, and migrations now run at startup.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_012CzABvBkbcFqoHdmi8b9WB
This commit is contained in:
thepraandClaude Opus 5.5 committed 2026-10-01 10:54:17 +02:00
1 parent 611abb5857
commit bf7c88ce71
13 files changed
+239 -4

No files matched your search

@@ -0,0 +1,80 @@
using AngleSharp.Css.Dom;
using AngleSharp.Dom;
using Ganss.Xss;
using System.Text.RegularExpressions;
namespace PrivaPub.Federation.Objects
{
public static partial class ContentSanitizer
{
static readonly HtmlSanitizer Sanitizer = Build();
public static string Html(string html) =>
string.IsNullOrWhiteSpace(html) ? string.Empty : Sanitizer.Sanitize(html).Trim();
static HtmlSanitizer Build()
{
var sanitizer = new HtmlSanitizer(new HtmlSanitizerOptions
{
AllowedTags = new HashSet<string>(StringComparer.OrdinalIgnoreCase)
{
"p", "br", "span", "a", "abbr", "del", "s", "pre", "blockquote", "code", "b", "strong", "u", "i", "em",
"sub", "sup", "ul", "ol", "li", "ruby", "rt", "rp", "h1", "h2", "h3", "h4", "h5", "h6"
},
AllowedAttributes = new HashSet<string>(StringComparer.OrdinalIgnoreCase)
{
"href", "rel", "class", "translate", "start", "reversed", "value", "title"
},
AllowedCssProperties = new HashSet<string>(),
AllowedAtRules = new HashSet<CssRuleType>(),
AllowedSchemes = new HashSet<string>(StringComparer.OrdinalIgnoreCase)
{
"http", "https", "dat", "dweb", "ipfs", "ipns", "ssb", "gopher", "xmpp", "magnet", "gemini"
},
UriAttributes = new HashSet<string>(StringComparer.OrdinalIgnoreCase) { "href" }
})
{
KeepChildNodes = true
};
foreach (var allowed in new[] { "mention", "hashtag", "ellipsis", "invisible" })
sanitizer.AllowedClasses.Add(allowed);
sanitizer.RemovingCssClass += (_, e) => e.Cancel = MicroformatClass().IsMatch(e.CssClass);
sanitizer.RemovingTag += (_, e) =>
{
if (e.Tag.LocalName is "script" or "style" or "template" or "iframe" or "object" or "embed" or "noscript" or "svg" or "math")
e.Tag.InnerHtml = string.Empty;
};
sanitizer.PostProcessNode += (_, e) =>
{
if (e.Node is not IElement element)
return;
switch (element.LocalName)
{
case "a":
if (!SchemePrefix().IsMatch(element.GetAttribute("href") ?? string.Empty))
element.RemoveAttribute("href");
element.SetAttribute("rel", "nofollow noopener noreferrer");
element.SetAttribute("target", "_blank");
break;
case "h1" or "h2" or "h3" or "h4" or "h5" or "h6":
var paragraph = e.Document.CreateElement("p");
var strong = e.Document.CreateElement("strong");
while (element.FirstChild != default)
strong.AppendChild(element.FirstChild);
paragraph.AppendChild(strong);
e.ReplacementNodes.Add(paragraph);
break;
}
};
return sanitizer;
}
[GeneratedRegex("^[a-z][a-z0-9+.-]*:", RegexOptions.IgnoreCase)]
private static partial Regex SchemePrefix();
[GeneratedRegex("^(h|p|u|dt|e)-[a-z0-9-]+$")]
private static partial Regex MicroformatClass();
}
}