Coverage for server / utilities / html_sanitizer.py: 83%

52 statements  

« prev     ^ index     » next       coverage.py v7.13.4, created at 2026-10-04 09:33 +0000

1""" 

2HTML sanitizer for rich-text question fields (Teacher-Student). 

3 

4Ported from mathmatterstx-services (EI-2481/EI-3260) for EI-3321. The staff bank has 

5sanitised every write path since EI-2481; the TEACHER bank had no sanitiser at all, so 

6a question could be stored with <script> and on*= handlers intact. The allow-list below 

7was already written against THIS repo's editor stack (TinyMCE + math-formula + 

8graph2d in eruditiontx-client-mvp), so it ports unchanged. 

9 

10Uses nh3 (Rust/ammonia bindings) to strip XSS vectors while preserving the 

11markup that the TinyMCE + math-formula + graph2d editor stack legitimately 

12produces. 

13 

14Allow-list rationale (sourced from TinyMCE editor config in 

15eruditiontx-client-mvp/src/utils/components/TinyMCETextEditor.jsx and the 

16math-formula plugin at 

17eruditiontx-client-mvp/src/lib/tinymce/plugins/math-formula/plugin.js): 

18 

19Tags 

20---- 

21Standard formatting : p, br, span, div, strong, b, em, i, u, s, sub, sup 

22Lists : ul, ol, li 

23Tables : table, thead, tbody, tr, td, th, colgroup, col 

24Media : img, a 

25Headings : h1-h6 

26Semantic : blockquote, code, pre 

27 

28Math / graph marks : 

29 - <span class="mfe-formula" data-latex="..." contenteditable="false"> … 

30 Rendered MathLive formula (mfe-formula CSS class, data-latex attribute). 

31 - <span class="graph2d-embed" data-graph2d="..." contenteditable="false"> … 

32 Embedded 2-D graph placeholder (graph2d-embed class, data-graph2d attribute). 

33 - <math-field> custom element emitted by MathLive in the editor dialog; 

34 nh3 strips the tag but preserves text content, which is acceptable because 

35 math-field elements only appear transiently in the editor dialog, NOT in 

36 the stored HTML. The math formula inserted into the TinyMCE body is always 

37 the rendered <span class="mfe-formula"> form. 

38 

39Attributes 

40---------- 

41Global (*): class, style, contenteditable + all data-* attributes (via 

42 generic_attribute_prefixes={'data-'}). The data-* prefix covers data-latex, 

43 data-graph2d, and any future plugin data attributes in one rule. 

44img : src, alt, width, height 

45a : href (javascript: / vbscript: URIs stripped by nh3's url_schemes) 

46td / th : colspan, rowspan 

47col : span 

48 

49Stripped 

50-------- 

51 - <script> (tag + content via clean_content_tags) 

52 - All on* event handler attributes (not in allow-list → dropped) 

53 - javascript: / vbscript: URIs on href/src (nh3 url_schemes; only safe 

54 schemes allowed: http, https, mailto, data — see URL_SCHEMES below) 

55 - Any tag not in ALLOWED_TAGS has its tag stripped but text preserved 

56 

57MathJax / LaTeX delimiter safety 

58--------------------------------- 

59nh3 operates only on HTML tags. LaTeX delimiters \\(...\\), \\[...\\], and 

60$$...$$ are plain text — they are never parsed as tags and pass through 

61unchanged. 

62""" 

63 

64import logging 

65import html 

66import re 

67 

68# nh3 (Rust/ammonia) is the preferred sanitizer. It is a compiled dependency, so 

69# if it isn't installed in the deployed environment we MUST NOT crash the app at 

70# import time (that took admin-staff QA down once). Fall back to a conservative 

71# stdlib stripper that still removes the primary XSS vectors. nh3 is declared in 

72# pyproject.toml so the full allow-list sanitizer is used in real deploys - 

73# requirements.txt is generated from uv.lock and must not be hand-edited. 

74try: 

75 import nh3 

76 

77 _NH3_AVAILABLE = True 

78except ImportError: # pragma: no cover - exercised only when nh3 is absent 

79 nh3 = None 

80 _NH3_AVAILABLE = False 

81 logging.getLogger(__name__).warning( 

82 "nh3 not installed — falling back to the stdlib HTML stripper. " 

83 "Install nh3 (see requirements.txt) for full allow-list sanitization." 

84 ) 

85 

86# Fallback (no-nh3) patterns: strip <script>/<style>/<noscript> blocks, all 

87# on*= event-handler attributes, and javascript:/vbscript: URIs. Less precise 

88# than the nh3 allow-list, but it neutralizes script execution. 

89_FALLBACK_CONTENT_TAGS_RE = re.compile( 

90 r"<(script|style|noscript)\b[^>]*>.*?</\1>", re.IGNORECASE | re.DOTALL 

91) 

92_FALLBACK_ONHANDLER_RE = re.compile( 

93 r"""\s+on\w+\s*=\s*(?:"[^"]*"|'[^']*'|[^\s>]+)""", re.IGNORECASE 

94) 

95_FALLBACK_JS_URI_RE = re.compile( 

96 r"""((?:href|src)\s*=\s*)(?:"\s*(?:javascript|vbscript):[^"]*"|'\s*(?:javascript|vbscript):[^']*')""", 

97 re.IGNORECASE, 

98) 

99 

100 

101def _fallback_sanitize(html: str) -> str: 

102 """Conservative stdlib sanitization used only when nh3 is unavailable.""" 

103 cleaned = _FALLBACK_CONTENT_TAGS_RE.sub("", html) 

104 cleaned = _FALLBACK_ONHANDLER_RE.sub("", cleaned) 

105 cleaned = _FALLBACK_JS_URI_RE.sub(r'\1""', cleaned) 

106 return cleaned 

107 

108 

109# --------------------------------------------------------------------------- 

110# HTML tag allow-list — mirrors TinyMCE's `plugins` config + math elements 

111# --------------------------------------------------------------------------- 

112_ALLOWED_TAGS: frozenset[str] = frozenset( 

113 { 

114 # Block formatting 

115 "p", 

116 "div", 

117 "br", 

118 "blockquote", 

119 "pre", 

120 "code", 

121 # Inline formatting 

122 "span", 

123 "strong", 

124 "b", 

125 "em", 

126 "i", 

127 "u", 

128 "s", 

129 "sub", 

130 "sup", 

131 # Lists 

132 "ul", 

133 "ol", 

134 "li", 

135 # Tables 

136 "table", 

137 "thead", 

138 "tbody", 

139 "tr", 

140 "td", 

141 "th", 

142 "colgroup", 

143 "col", 

144 # Media 

145 "img", 

146 "a", 

147 # Headings 

148 "h1", 

149 "h2", 

150 "h3", 

151 "h4", 

152 "h5", 

153 "h6", 

154 } 

155) 

156 

157# --------------------------------------------------------------------------- 

158# Per-element attribute allow-list. 

159# data-* attributes are covered by generic_attribute_prefixes below. 

160# --------------------------------------------------------------------------- 

161_ALLOWED_ATTRS: dict[str, set[str]] = { 

162 # Global attributes allowed on any element 

163 "*": {"class", "style", "contenteditable"}, 

164 # img: allow src/alt/dimensions 

165 "img": {"src", "alt", "width", "height"}, 

166 # a: href is allowed; javascript:/vbscript: are blocked by url_schemes 

167 "a": {"href", "class", "style"}, 

168 # Table cell spanning 

169 "td": {"colspan", "rowspan", "class", "style"}, 

170 "th": {"colspan", "rowspan", "class", "style"}, 

171 # Colgroup column width hints 

172 "col": {"span", "style"}, 

173} 

174 

175# --------------------------------------------------------------------------- 

176# URL schemes permitted in href / src attributes. 

177# This blocks javascript: / vbscript: / data: on links while allowing data: 

178# URIs on images (TinyMCE may embed images as data: URLs during editing). 

179# --------------------------------------------------------------------------- 

180_URL_SCHEMES: frozenset[str] = frozenset({"http", "https", "mailto", "data"}) 

181 

182# Tags whose *content* should also be stripped (not just the tag itself). 

183_STRIP_CONTENT_TAGS: frozenset[str] = frozenset({"script", "style", "noscript"}) 

184 

185 

186def strip_html(text: str | None) -> str | None: 

187 """ 

188 Strip ALL HTML tags from a plain-text field (title, description, tag). 

189 

190 Use this for metadata fields that are stored as plain text and must never 

191 contain markup. The result is the visible text content with every tag 

192 removed. 

193 

194 Args: 

195 text: Raw string that may contain HTML, or None. 

196 

197 Returns: 

198 Plain text with all tags stripped, or None if the input was None. 

199 An empty string input returns an empty string. 

200 """ 

201 if text is None: 

202 return None 

203 if not text: 

204 return text 

205 

206 if _NH3_AVAILABLE: 

207 # tags=set() → every tag is stripped; attributes={} → no attrs kept. 

208 return nh3.clean(text, tags=set(), attributes={}) 

209 

210 # Fallback: run the conservative cleaner then strip remaining tags with a 

211 # simple regex (good enough for plain-text fields when nh3 is absent). 

212 cleaned = _fallback_sanitize(text) 

213 return re.sub(r"<[^>]+>", "", cleaned) 

214 

215 

216def sanitize_rich_text(html: str | None) -> str | None: 

217 """ 

218 Sanitize a rich-text HTML string produced by the TinyMCE + math editor. 

219 

220 Strips XSS vectors (on* attributes, <script>, javascript: / vbscript: URIs) 

221 while preserving formatting tags, math markup (mfe-formula spans, data-latex), 

222 graph embeds (graph2d-embed spans, data-graph2d), and MathJax LaTeX 

223 delimiters (which are plain text and are never touched by the sanitizer). 

224 

225 Args: 

226 html: Raw HTML string from the editor, or None / empty. 

227 

228 Returns: 

229 Sanitized HTML string, or None if the input was None. 

230 An empty string input returns an empty string. 

231 """ 

232 if html is None: 

233 return None 

234 if not html: 

235 return html 

236 

237 if not _NH3_AVAILABLE: 

238 return _fallback_sanitize(html) 

239 

240 return nh3.clean( 

241 html, 

242 tags=_ALLOWED_TAGS, 

243 clean_content_tags=_STRIP_CONTENT_TAGS, 

244 attributes=_ALLOWED_ATTRS, 

245 # Allows all data-* attributes: data-latex, data-graph2d, etc. 

246 generic_attribute_prefixes={"data-"}, 

247 strip_comments=True, 

248 url_schemes=_URL_SCHEMES, 

249 # nh3 default: add rel="noopener noreferrer" to links (keep) 

250 link_rel="noopener noreferrer", 

251 ) 

252 

253 

254def sanitize_answer_value(value): 

255 """Sanitise every string inside a student answer, whatever shape it takes. 

256 

257 A student answer is `str | list[str | dict] | dict` — a single-string rule 

258 would miss the blanks of a drop-down or drag-and-drop answer. Non-string 

259 leaves (numbers, booleans, None) are returned untouched. 

260 

261 Uses `sanitize_rich_text` rather than `strip_html` on purpose: answers come 

262 from TinyMCEAnswerEditor and a typed formula is stored as 

263 `<span class="mfe-formula" data-latex="...">`, which grading collapses to its 

264 LaTeX. Stripping would destroy both the formatting and the maths. 

265 

266 Developer: Allan Ninal — 2026-09-23 (EI-T717 / EI-T721) 

267 """ 

268 if isinstance(value, str): 

269 return sanitize_rich_text(value) 

270 if isinstance(value, list): 

271 return [sanitize_answer_value(item) for item in value] 

272 if isinstance(value, dict): 

273 return {key: sanitize_answer_value(item) for key, item in value.items()} 

274 return value 

275 

276 

277def to_plain_text(text: str | None) -> str | None: 

278 """Strip markup for a plain-text field WITHOUT mangling ordinary maths text. 

279 

280 `strip_html` removes tags but HTML-escapes whatever survives, so a teacher 

281 typing ``solve x < 5`` into a class announcement would have it stored as 

282 ``solve x &lt; 5`` and shown with the entity visible. On a maths product these 

283 fields (class title/description, class announcements) are exactly where ``<`` 

284 and ``>`` legitimately appear, so that is a real regression. 

285 

286 After `strip_html` has run, every remaining entity is text that nh3 already 

287 judged NOT to be markup — so putting it back is safe, with one exception: an 

288 input that arrived ALREADY entity-encoded (``&lt;script&gt;``) would unescape 

289 into a live tag. The test for that is direct rather than pattern-matched: if 

290 re-cleaning the unescaped candidate REMOVES anything, unescaping reconstructed 

291 markup, so the escaped form is kept. 

292 

293 "a < b" -> nothing removed -> kept readable as "a < b" 

294 "&lt;script&gt;alert()" -> "<script>…" is removed -> stays escaped 

295 

296 Returns None for None, mirroring `strip_html`. 

297 """ 

298 if text is None: 

299 return None 

300 stripped = strip_html(text) or "" 

301 candidate = html.unescape(stripped) 

302 recleaned = html.unescape(strip_html(candidate) or "") 

303 if recleaned != candidate: 

304 return stripped 

305 return candidate