- sanitizer XSS serveur (markdown + page de partage) [BUG-021/022] - rate-limit/lockout MFA [BUG-023] - isolation vaults par segments [BUG-024] - caps regex ReDoS [BUG-025] - SSRF webhooks + secrets externalises [BUG-026] - rotation/revocation des jetons [BUG-027] - politique de mot de passe + invalidation sessions [BUG-028] - verrous users.json [BUG-029] - IP reelle dans les audits [BUG-030] - rate-limit par compte [BUG-031] - symlinks hors vault ignores [BUG-032] - recherche simple via inverted index [BUG-033] - token en memoire + cookie HttpOnly, CSP durcie [BUG-034] Tests: pytest 961 passed / 6 skipped, ruff 0, mypy 0, frontend vert.
287 lines
11 KiB
Python
287 lines
11 KiB
Python
"""Whitelist HTML sanitizer used for untrusted markdown / AI output.
|
|
|
|
The markdown renderer runs with ``escape=False`` so raw HTML authored inside a
|
|
vault (or returned by a model) reaches the browser. This module scrubs the
|
|
rendered HTML against a strict whitelist of tags and attributes, drops
|
|
dangerous URL schemes and strips every event handler / ``style`` attribute.
|
|
|
|
Implemented with the standard library only (no third-party dependency) so the
|
|
runtime footprint stays unchanged. It is *not* a full HTML5 parser: it is a
|
|
conservative, allow-list based filter intended for already well-formed output
|
|
produced by mistune and the image/wikilink pre-processors.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import html as _html
|
|
from html.parser import HTMLParser
|
|
|
|
__all__ = ["is_safe_url", "sanitize_html"]
|
|
|
|
# Tags whose *content* is discarded entirely (never rendered as text).
|
|
_DROP_CONTENT_TAGS = frozenset({
|
|
"script", "style", "iframe", "object", "embed", "template", "noscript",
|
|
"svg", "math", "applet", "form", "button", "select", "textarea", "option",
|
|
"frame", "frameset", "base", "link", "meta", "title", "head",
|
|
})
|
|
|
|
# Tags kept in the output (text content preserved for unknown tags).
|
|
_ALLOWED_TAGS = frozenset({
|
|
"a", "abbr", "b", "blockquote", "br", "caption", "code", "col", "colgroup",
|
|
"dd", "del", "details", "div", "dl", "dt", "em", "figcaption", "figure",
|
|
"h1", "h2", "h3", "h4", "h5", "h6", "hr", "i", "img", "input", "kbd", "li",
|
|
"mark", "ol", "p", "pre", "q", "s", "section", "small", "span", "strong",
|
|
"sub", "summary", "sup", "table", "tbody", "td", "tfoot", "th", "thead",
|
|
"time", "tr", "u", "ul", "video", "audio", "source", "track",
|
|
})
|
|
|
|
# Attributes allowed on any element.
|
|
_GLOBAL_ATTRS = frozenset({"class", "id", "title", "dir", "lang", "role"})
|
|
|
|
# Per-tag attribute whitelist (in addition to globals and ``data-*``).
|
|
_TAG_ATTRS: dict[str, frozenset[str]] = {
|
|
"a": frozenset({"href", "target", "rel", "name", "download"}),
|
|
"img": frozenset({"src", "alt", "width", "height", "loading"}),
|
|
"input": frozenset({"type", "checked", "disabled", "value"}),
|
|
"ol": frozenset({"start", "type", "reversed"}),
|
|
"ul": frozenset({"type"}),
|
|
"li": frozenset({"value"}),
|
|
"td": frozenset({"colspan", "rowspan", "align", "valign"}),
|
|
"th": frozenset({"colspan", "rowspan", "align", "valign", "scope"}),
|
|
"col": frozenset({"span", "width"}),
|
|
"colgroup": frozenset({"span"}),
|
|
"video": frozenset({"src", "controls", "width", "height", "loop", "muted",
|
|
"poster", "preload", "playsinline"}),
|
|
"audio": frozenset({"src", "controls", "loop", "muted", "preload"}),
|
|
"source": frozenset({"src", "type", "srcset", "media"}),
|
|
"track": frozenset({"src", "kind", "srclang", "label", "default"}),
|
|
"details": frozenset({"open"}),
|
|
"time": frozenset({"datetime"}),
|
|
"blockquote": frozenset({"cite"}),
|
|
"q": frozenset({"cite"}),
|
|
}
|
|
|
|
# URL-bearing attributes per tag, and whether ``data:`` URIs are acceptable.
|
|
_URL_ATTRS: dict[str, frozenset[str]] = {
|
|
"a": frozenset({"href"}),
|
|
"img": frozenset({"src"}),
|
|
"video": frozenset({"src", "poster"}),
|
|
"audio": frozenset({"src"}),
|
|
"source": frozenset({"src", "srcset"}),
|
|
"track": frozenset({"src"}),
|
|
"blockquote": frozenset({"cite"}),
|
|
"q": frozenset({"cite"}),
|
|
}
|
|
|
|
_SAFE_SCHEMES = frozenset({"http", "https", "mailto", "tel", "ftp"})
|
|
|
|
# Characters that browsers ignore inside a scheme (tab/newline/CR) and that
|
|
# could otherwise smuggle ``java\tscript:`` past a naive check.
|
|
_URL_STRIP_CHARS = "\t\n\r\x00"
|
|
|
|
|
|
def is_safe_url(value: str, *, allow_data: bool = False, tag: str = "") -> bool:
|
|
"""Return True when *value* is a URL with an allowed scheme.
|
|
|
|
Relative URLs (``/foo``, ``./foo``, ``#anchor``) are allowed. Dangerous
|
|
schemes such as ``javascript:`` and ``vbscript:`` are always rejected.
|
|
``data:`` URIs are only allowed for image/video/audio sources.
|
|
"""
|
|
if value is None:
|
|
return False
|
|
# Decode entities and strip whitespace/control chars before inspecting.
|
|
candidate = _html.unescape(str(value)).strip()
|
|
for ch in _URL_STRIP_CHARS:
|
|
candidate = candidate.replace(ch, "")
|
|
if not candidate:
|
|
return False
|
|
|
|
# Detect a scheme: ``scheme:`` where scheme is [a-zA-Z][a-zA-Z0-9+.-]*
|
|
lowered = candidate.lower()
|
|
if lowered.startswith("data:"):
|
|
if not allow_data:
|
|
return False
|
|
# Only media data URIs are permitted.
|
|
if tag in ("img",):
|
|
return lowered.startswith("data:image/")
|
|
if tag in ("video", "audio", "source", "track"):
|
|
return (
|
|
lowered.startswith("data:image/")
|
|
or lowered.startswith("data:video/")
|
|
or lowered.startswith("data:audio/")
|
|
)
|
|
return False
|
|
|
|
if lowered.startswith("blob:"):
|
|
return tag in ("img", "video", "audio", "source")
|
|
|
|
# No scheme at all (relative / fragment / protocol-relative) → safe.
|
|
colon = candidate.find(":")
|
|
slash = candidate.find("/")
|
|
if colon == -1 or (slash != -1 and slash < colon):
|
|
return True
|
|
# ``//host`` protocol-relative has no scheme.
|
|
if candidate.startswith("//"):
|
|
return True
|
|
|
|
scheme = lowered[:colon]
|
|
if not scheme or not scheme[0].isalpha():
|
|
return True # not a real scheme, treat as relative
|
|
return scheme in _SAFE_SCHEMES
|
|
|
|
|
|
class _Sanitizer(HTMLParser):
|
|
"""Rebuild HTML while dropping anything not explicitly allowed."""
|
|
|
|
def __init__(self) -> None:
|
|
super().__init__(convert_charrefs=True)
|
|
self._out: list[str] = []
|
|
# Stack of tag names currently open (only allowed tags).
|
|
self._open: list[str] = []
|
|
# Stack tracking dropped-content depth: each entry is the tag name.
|
|
self._suppress: list[str] = []
|
|
|
|
# -- helpers ----------------------------------------------------------
|
|
def _filter_attrs(self, tag: str, attrs: list[tuple[str, str | None]]) -> str:
|
|
allowed_extra = _TAG_ATTRS.get(tag, frozenset())
|
|
url_attrs = _URL_ATTRS.get(tag, frozenset())
|
|
parts: list[str] = []
|
|
seen: set[str] = set()
|
|
for name, value in attrs:
|
|
if value is None:
|
|
value = ""
|
|
lname = name.lower()
|
|
if lname in seen:
|
|
continue
|
|
seen.add(lname)
|
|
# Event handlers and style are never allowed.
|
|
if lname.startswith("on") or lname in ("style", "srcdoc", "formaction", "xlink:href"):
|
|
continue
|
|
if lname.startswith("data-") or lname.startswith("aria-"):
|
|
pass
|
|
elif lname not in _GLOBAL_ATTRS and lname not in allowed_extra:
|
|
continue
|
|
|
|
if lname in url_attrs:
|
|
allow_data = tag in ("img", "video", "audio", "source", "track")
|
|
# ``srcset`` may contain multiple comma-separated candidates.
|
|
if lname == "srcset":
|
|
if not _safe_srcset(value, tag):
|
|
continue
|
|
elif not is_safe_url(value, allow_data=allow_data, tag=tag):
|
|
continue
|
|
parts.append(f' {lname}="{_html.escape(value, quote=True)}"')
|
|
return "".join(parts)
|
|
|
|
def _emit_start(self, tag: str, attrs, self_closing: bool) -> None:
|
|
attrs_html = self._filter_attrs(tag, attrs)
|
|
if self_closing or tag in ("br", "hr", "img", "input", "col", "source", "track"):
|
|
self._out.append(f"<{tag}{attrs_html} />")
|
|
else:
|
|
self._out.append(f"<{tag}{attrs_html}>")
|
|
self._open.append(tag)
|
|
|
|
# -- HTMLParser callbacks --------------------------------------------
|
|
def handle_starttag(self, tag: str, attrs) -> None:
|
|
tag = tag.lower()
|
|
if tag in _DROP_CONTENT_TAGS:
|
|
self._suppress.append(tag)
|
|
return
|
|
if self._suppress:
|
|
return
|
|
if tag not in _ALLOWED_TAGS:
|
|
return # drop the tag, keep its text content
|
|
self._emit_start(tag, attrs, self_closing=False)
|
|
|
|
def handle_startendtag(self, tag: str, attrs) -> None:
|
|
tag = tag.lower()
|
|
if tag in _DROP_CONTENT_TAGS or self._suppress:
|
|
return
|
|
if tag not in _ALLOWED_TAGS:
|
|
return
|
|
self._emit_start(tag, attrs, self_closing=True)
|
|
|
|
def handle_endtag(self, tag: str) -> None:
|
|
tag = tag.lower()
|
|
if tag in _DROP_CONTENT_TAGS:
|
|
# Close the innermost matching suppress marker.
|
|
for i in range(len(self._suppress) - 1, -1, -1):
|
|
if self._suppress[i] == tag:
|
|
del self._suppress[i:]
|
|
break
|
|
return
|
|
if self._suppress:
|
|
return
|
|
if tag not in _ALLOWED_TAGS:
|
|
return
|
|
# Only close if currently open (tolerate malformed nesting).
|
|
if tag in self._open:
|
|
while self._open:
|
|
top = self._open.pop()
|
|
self._out.append(f"</{top}>")
|
|
if top == tag:
|
|
break
|
|
|
|
def handle_data(self, data: str) -> None:
|
|
if self._suppress:
|
|
return
|
|
self._out.append(_html.escape(data, quote=False))
|
|
|
|
def handle_comment(self, data: str) -> None:
|
|
return # comments are dropped
|
|
|
|
def handle_decl(self, decl: str) -> None:
|
|
return
|
|
|
|
def handle_pi(self, data: str) -> None:
|
|
return
|
|
|
|
def handle_entityref(self, name: str) -> None:
|
|
if self._suppress:
|
|
return
|
|
self._out.append(f"&{name};")
|
|
|
|
def handle_charref(self, name: str) -> None:
|
|
if self._suppress:
|
|
return
|
|
self._out.append(f"&#{name};")
|
|
|
|
def get_html(self) -> str:
|
|
# Close any tags left open by malformed input.
|
|
while self._open:
|
|
self._out.append(f"</{self._open.pop()}>")
|
|
return "".join(self._out)
|
|
|
|
|
|
def _safe_srcset(value: str, tag: str) -> bool:
|
|
"""Validate every candidate in a ``srcset`` attribute."""
|
|
for candidate in value.split(","):
|
|
candidate = candidate.strip()
|
|
if not candidate:
|
|
continue
|
|
url = candidate.split()[0] if candidate.split() else candidate
|
|
if not is_safe_url(url, allow_data=(tag == "img"), tag=tag):
|
|
return False
|
|
return True
|
|
|
|
|
|
def sanitize_html(html: str) -> str:
|
|
"""Return *html* with only whitelisted tags/attributes/schemes preserved.
|
|
|
|
Args:
|
|
html: Untrusted HTML (typically mistune output with ``escape=False``).
|
|
|
|
Returns:
|
|
Sanitized HTML string.
|
|
"""
|
|
if not html:
|
|
return ""
|
|
parser = _Sanitizer()
|
|
try:
|
|
parser.feed(html)
|
|
parser.close()
|
|
except Exception:
|
|
# Never let sanitization crash a request; fail closed to plain text.
|
|
return _html.escape(html)
|
|
return parser.get_html()
|