Coverage for server / utilities / html_sanitizer.py: 83%
52 statements
« prev ^ index » next coverage.py v7.13.4, created at 2026-10-04 09:33 +0000
« prev ^ index » next coverage.py v7.13.4, created at 2026-10-04 09:33 +0000
1"""
2HTML sanitizer for rich-text question fields (Teacher-Student).
4Ported from mathmatterstx-services (EI-2481/EI-3260) for EI-3321. The staff bank has
5sanitised every write path since EI-2481; the TEACHER bank had no sanitiser at all, so
6a question could be stored with <script> and on*= handlers intact. The allow-list below
7was already written against THIS repo's editor stack (TinyMCE + math-formula +
8graph2d in eruditiontx-client-mvp), so it ports unchanged.
10Uses nh3 (Rust/ammonia bindings) to strip XSS vectors while preserving the
11markup that the TinyMCE + math-formula + graph2d editor stack legitimately
12produces.
14Allow-list rationale (sourced from TinyMCE editor config in
15eruditiontx-client-mvp/src/utils/components/TinyMCETextEditor.jsx and the
16math-formula plugin at
17eruditiontx-client-mvp/src/lib/tinymce/plugins/math-formula/plugin.js):
19Tags
20----
21Standard formatting : p, br, span, div, strong, b, em, i, u, s, sub, sup
22Lists : ul, ol, li
23Tables : table, thead, tbody, tr, td, th, colgroup, col
24Media : img, a
25Headings : h1-h6
26Semantic : blockquote, code, pre
28Math / graph marks :
29 - <span class="mfe-formula" data-latex="..." contenteditable="false"> …
30 Rendered MathLive formula (mfe-formula CSS class, data-latex attribute).
31 - <span class="graph2d-embed" data-graph2d="..." contenteditable="false"> …
32 Embedded 2-D graph placeholder (graph2d-embed class, data-graph2d attribute).
33 - <math-field> custom element emitted by MathLive in the editor dialog;
34 nh3 strips the tag but preserves text content, which is acceptable because
35 math-field elements only appear transiently in the editor dialog, NOT in
36 the stored HTML. The math formula inserted into the TinyMCE body is always
37 the rendered <span class="mfe-formula"> form.
39Attributes
40----------
41Global (*): class, style, contenteditable + all data-* attributes (via
42 generic_attribute_prefixes={'data-'}). The data-* prefix covers data-latex,
43 data-graph2d, and any future plugin data attributes in one rule.
44img : src, alt, width, height
45a : href (javascript: / vbscript: URIs stripped by nh3's url_schemes)
46td / th : colspan, rowspan
47col : span
49Stripped
50--------
51 - <script> (tag + content via clean_content_tags)
52 - All on* event handler attributes (not in allow-list → dropped)
53 - javascript: / vbscript: URIs on href/src (nh3 url_schemes; only safe
54 schemes allowed: http, https, mailto, data — see URL_SCHEMES below)
55 - Any tag not in ALLOWED_TAGS has its tag stripped but text preserved
57MathJax / LaTeX delimiter safety
58---------------------------------
59nh3 operates only on HTML tags. LaTeX delimiters \\(...\\), \\[...\\], and
60$$...$$ are plain text — they are never parsed as tags and pass through
61unchanged.
62"""
64import logging
65import html
66import re
68# nh3 (Rust/ammonia) is the preferred sanitizer. It is a compiled dependency, so
69# if it isn't installed in the deployed environment we MUST NOT crash the app at
70# import time (that took admin-staff QA down once). Fall back to a conservative
71# stdlib stripper that still removes the primary XSS vectors. nh3 is declared in
72# pyproject.toml so the full allow-list sanitizer is used in real deploys -
73# requirements.txt is generated from uv.lock and must not be hand-edited.
74try:
75 import nh3
77 _NH3_AVAILABLE = True
78except ImportError: # pragma: no cover - exercised only when nh3 is absent
79 nh3 = None
80 _NH3_AVAILABLE = False
81 logging.getLogger(__name__).warning(
82 "nh3 not installed — falling back to the stdlib HTML stripper. "
83 "Install nh3 (see requirements.txt) for full allow-list sanitization."
84 )
86# Fallback (no-nh3) patterns: strip <script>/<style>/<noscript> blocks, all
87# on*= event-handler attributes, and javascript:/vbscript: URIs. Less precise
88# than the nh3 allow-list, but it neutralizes script execution.
89_FALLBACK_CONTENT_TAGS_RE = re.compile(
90 r"<(script|style|noscript)\b[^>]*>.*?</\1>", re.IGNORECASE | re.DOTALL
91)
92_FALLBACK_ONHANDLER_RE = re.compile(
93 r"""\s+on\w+\s*=\s*(?:"[^"]*"|'[^']*'|[^\s>]+)""", re.IGNORECASE
94)
95_FALLBACK_JS_URI_RE = re.compile(
96 r"""((?:href|src)\s*=\s*)(?:"\s*(?:javascript|vbscript):[^"]*"|'\s*(?:javascript|vbscript):[^']*')""",
97 re.IGNORECASE,
98)
101def _fallback_sanitize(html: str) -> str:
102 """Conservative stdlib sanitization used only when nh3 is unavailable."""
103 cleaned = _FALLBACK_CONTENT_TAGS_RE.sub("", html)
104 cleaned = _FALLBACK_ONHANDLER_RE.sub("", cleaned)
105 cleaned = _FALLBACK_JS_URI_RE.sub(r'\1""', cleaned)
106 return cleaned
109# ---------------------------------------------------------------------------
110# HTML tag allow-list — mirrors TinyMCE's `plugins` config + math elements
111# ---------------------------------------------------------------------------
112_ALLOWED_TAGS: frozenset[str] = frozenset(
113 {
114 # Block formatting
115 "p",
116 "div",
117 "br",
118 "blockquote",
119 "pre",
120 "code",
121 # Inline formatting
122 "span",
123 "strong",
124 "b",
125 "em",
126 "i",
127 "u",
128 "s",
129 "sub",
130 "sup",
131 # Lists
132 "ul",
133 "ol",
134 "li",
135 # Tables
136 "table",
137 "thead",
138 "tbody",
139 "tr",
140 "td",
141 "th",
142 "colgroup",
143 "col",
144 # Media
145 "img",
146 "a",
147 # Headings
148 "h1",
149 "h2",
150 "h3",
151 "h4",
152 "h5",
153 "h6",
154 }
155)
157# ---------------------------------------------------------------------------
158# Per-element attribute allow-list.
159# data-* attributes are covered by generic_attribute_prefixes below.
160# ---------------------------------------------------------------------------
161_ALLOWED_ATTRS: dict[str, set[str]] = {
162 # Global attributes allowed on any element
163 "*": {"class", "style", "contenteditable"},
164 # img: allow src/alt/dimensions
165 "img": {"src", "alt", "width", "height"},
166 # a: href is allowed; javascript:/vbscript: are blocked by url_schemes
167 "a": {"href", "class", "style"},
168 # Table cell spanning
169 "td": {"colspan", "rowspan", "class", "style"},
170 "th": {"colspan", "rowspan", "class", "style"},
171 # Colgroup column width hints
172 "col": {"span", "style"},
173}
175# ---------------------------------------------------------------------------
176# URL schemes permitted in href / src attributes.
177# This blocks javascript: / vbscript: / data: on links while allowing data:
178# URIs on images (TinyMCE may embed images as data: URLs during editing).
179# ---------------------------------------------------------------------------
180_URL_SCHEMES: frozenset[str] = frozenset({"http", "https", "mailto", "data"})
182# Tags whose *content* should also be stripped (not just the tag itself).
183_STRIP_CONTENT_TAGS: frozenset[str] = frozenset({"script", "style", "noscript"})
186def strip_html(text: str | None) -> str | None:
187 """
188 Strip ALL HTML tags from a plain-text field (title, description, tag).
190 Use this for metadata fields that are stored as plain text and must never
191 contain markup. The result is the visible text content with every tag
192 removed.
194 Args:
195 text: Raw string that may contain HTML, or None.
197 Returns:
198 Plain text with all tags stripped, or None if the input was None.
199 An empty string input returns an empty string.
200 """
201 if text is None:
202 return None
203 if not text:
204 return text
206 if _NH3_AVAILABLE:
207 # tags=set() → every tag is stripped; attributes={} → no attrs kept.
208 return nh3.clean(text, tags=set(), attributes={})
210 # Fallback: run the conservative cleaner then strip remaining tags with a
211 # simple regex (good enough for plain-text fields when nh3 is absent).
212 cleaned = _fallback_sanitize(text)
213 return re.sub(r"<[^>]+>", "", cleaned)
216def sanitize_rich_text(html: str | None) -> str | None:
217 """
218 Sanitize a rich-text HTML string produced by the TinyMCE + math editor.
220 Strips XSS vectors (on* attributes, <script>, javascript: / vbscript: URIs)
221 while preserving formatting tags, math markup (mfe-formula spans, data-latex),
222 graph embeds (graph2d-embed spans, data-graph2d), and MathJax LaTeX
223 delimiters (which are plain text and are never touched by the sanitizer).
225 Args:
226 html: Raw HTML string from the editor, or None / empty.
228 Returns:
229 Sanitized HTML string, or None if the input was None.
230 An empty string input returns an empty string.
231 """
232 if html is None:
233 return None
234 if not html:
235 return html
237 if not _NH3_AVAILABLE:
238 return _fallback_sanitize(html)
240 return nh3.clean(
241 html,
242 tags=_ALLOWED_TAGS,
243 clean_content_tags=_STRIP_CONTENT_TAGS,
244 attributes=_ALLOWED_ATTRS,
245 # Allows all data-* attributes: data-latex, data-graph2d, etc.
246 generic_attribute_prefixes={"data-"},
247 strip_comments=True,
248 url_schemes=_URL_SCHEMES,
249 # nh3 default: add rel="noopener noreferrer" to links (keep)
250 link_rel="noopener noreferrer",
251 )
254def sanitize_answer_value(value):
255 """Sanitise every string inside a student answer, whatever shape it takes.
257 A student answer is `str | list[str | dict] | dict` — a single-string rule
258 would miss the blanks of a drop-down or drag-and-drop answer. Non-string
259 leaves (numbers, booleans, None) are returned untouched.
261 Uses `sanitize_rich_text` rather than `strip_html` on purpose: answers come
262 from TinyMCEAnswerEditor and a typed formula is stored as
263 `<span class="mfe-formula" data-latex="...">`, which grading collapses to its
264 LaTeX. Stripping would destroy both the formatting and the maths.
266 Developer: Allan Ninal — 2026-09-23 (EI-T717 / EI-T721)
267 """
268 if isinstance(value, str):
269 return sanitize_rich_text(value)
270 if isinstance(value, list):
271 return [sanitize_answer_value(item) for item in value]
272 if isinstance(value, dict):
273 return {key: sanitize_answer_value(item) for key, item in value.items()}
274 return value
277def to_plain_text(text: str | None) -> str | None:
278 """Strip markup for a plain-text field WITHOUT mangling ordinary maths text.
280 `strip_html` removes tags but HTML-escapes whatever survives, so a teacher
281 typing ``solve x < 5`` into a class announcement would have it stored as
282 ``solve x < 5`` and shown with the entity visible. On a maths product these
283 fields (class title/description, class announcements) are exactly where ``<``
284 and ``>`` legitimately appear, so that is a real regression.
286 After `strip_html` has run, every remaining entity is text that nh3 already
287 judged NOT to be markup — so putting it back is safe, with one exception: an
288 input that arrived ALREADY entity-encoded (``<script>``) would unescape
289 into a live tag. The test for that is direct rather than pattern-matched: if
290 re-cleaning the unescaped candidate REMOVES anything, unescaping reconstructed
291 markup, so the escaped form is kept.
293 "a < b" -> nothing removed -> kept readable as "a < b"
294 "<script>alert()" -> "<script>…" is removed -> stays escaped
296 Returns None for None, mirroring `strip_html`.
297 """
298 if text is None:
299 return None
300 stripped = strip_html(text) or ""
301 candidate = html.unescape(stripped)
302 recleaned = html.unescape(strip_html(candidate) or "")
303 if recleaned != candidate:
304 return stripped
305 return candidate