Coverage for server / services / common / answer_checking.py: 96%

111 statements  

« prev     ^ index     » next       coverage.py v7.13.4, created at 2026-10-04 09:33 +0000

1"""Canonical correct-answer checking, shared across every question type. 

2 

3Single source of truth for "is this submitted answer correct" — the student 

4assignment-submit/scoring path and teacher-side reporting (Item Analysis, 

5live submission view) both call `is_answer_correct` so a submission is never 

6graded one way by the gradebook and reported a different way in analytics. 

7""" 

8import html 

9import re 

10 

11from server.utilities.graph_data_checker import ( 

12 grade_graph_answer, 

13 grade_interactive_dots_answer, 

14 is_graph_question_type, 

15 is_interactive_dots_question_type, 

16) 

17 

18 

19# Matches a math-formula span as inserted by the TinyMCE math-formula plugin: 

20# <span class="mfe-formula" data-latex="...">...rendered markup...</span>. 

21# Used to canonicalize a free-response answer for grading — see 

22# normalize_free_response_answer. 

23MFE_FORMULA_RE = re.compile( 

24 r'<span[^>]*class="mfe-formula"[^>]*data-latex="([^"]*)"[^>]*>.*?</span>', 

25 re.DOTALL, 

26) 

27 

28 

29def normalize_free_response_answer(value) -> str: 

30 """ 

31 Canonicalizes a free-response answer for comparison. 

32 

33 The stored value is TinyMCE HTML (the answer/model-answer editor — 

34 see FreeResponse.jsx/TinyMCEAnswerEditor.jsx in the client — wraps a 

35 typed math expression in `<span class="mfe-formula" data-latex="..."> 

36 ...rendered markup...</span>`), or, for older submissions made before 

37 that editor existed, a plain LaTeX string typed into a math-only 

38 field. A byte-for-byte comparison of the raw HTML is not a 

39 meaningful equality check the way it is for a chosen multiple-choice 

40 option (which round-trips the exact stored choice string verbatim): 

41 two independently-typed answers containing "the same" formula do not 

42 produce identical rendered markup, and a plain LaTeX string can never 

43 equal an HTML string at all. 

44 

45 Each math-formula span is collapsed to its `data-latex` source 

46 (stable across renders/typing sessions) instead of its rendered 

47 markup, remaining HTML tags are stripped, entities are decoded, and 

48 whitespace is collapsed — so paragraph/line-break/formatting 

49 differences that don't change the visible answer don't cause a 

50 false negative. 

51 """ 

52 if isinstance(value, dict): 

53 value = value.get("answer", "") 

54 if not isinstance(value, str): 

55 value = "" if value is None else str(value) 

56 

57 value = MFE_FORMULA_RE.sub(lambda m: m.group(1), value) 

58 value = re.sub(r"<[^>]+>", " ", value) 

59 value = html.unescape(value) 

60 return re.sub(r"\s+", " ", value).strip() 

61 

62 

63# Question types where each {id, answer} entry is a distinct blank/position 

64# (a dropdown, a drop target) rather than an independently-correct option in 

65# an unordered set (checkbox). Grading these must compare answers id-to-id, 

66# not as a bag of texts — see _build_id_answer_map. "drop-down" is the 

67# Single-Stimulus "drop-down" group type (group.get("type")) — distinct 

68# from the standalone "drop-down-menu" question type, but the same 

69# id-per-blank shape (see SingleStimulusEditor.jsx / DropdownMenuV2.jsx 

70# on the client), so it gets the same position-sensitive fallback. 

71# "grid-question" reuses this same id-to-id comparison unchanged: each row 

72# is a "blank" (id = row id, answer = the chosen column's text) — see 

73# docs/features/grid-question.md. 

74_POSITION_SENSITIVE_TYPES = frozenset({"drop-down-menu", "drag-and-drop", "drop-down", "grid-question"}) 

75 

76 

77def _normalize_question_type(question_type) -> str | None: 

78 if question_type is None: 

79 return None 

80 return str(question_type).strip().lower().replace("_", "-") 

81 

82 

83_GROUP_GRADED_TYPES = frozenset({"single-stimulus", "multi-part-question"}) 

84 

85 

86def is_multi_group_type(question_type) -> bool: 

87 """Whether `question_type` is graded group-by-group (Single-Stimulus, 

88 Multi-Part-Question) rather than as one flat answer. Both share the same 

89 `groups` shape and grading — Multi-Part-Question's only difference is an 

90 authoring-time-only per-group `questionText` field this module never 

91 reads.""" 

92 return _normalize_question_type(question_type) in _GROUP_GRADED_TYPES 

93 

94 

95def grade_single_stimulus_answer(student_answer, groups) -> dict: 

96 """Grade a Single-Stimulus answer — each group scored independently. 

97 

98 Args: 

99 student_answer: the combined answer for the question, shaped 

100 ``{group_id: <that group's answer, same shape its own `type` 

101 normally submits>}`` (or anything else / missing for unanswered). 

102 groups: the question document's `groups` list — each carries its own 

103 `type`/`choices`/`correctAnswer`/`points`. 

104 

105 Returns a dict with `is_correct` (all groups correct), `earned_points`, 

106 `max_points`, and `group_results` (per-group `{isCorrect, earnedPoints, 

107 maxPoints}`, keyed by group_id) — a correct answer in one group never 

108 affects another's score. 

109 """ 

110 answers_by_group = student_answer if isinstance(student_answer, dict) else {} 

111 group_results: dict = {} 

112 earned_points = 0 

113 max_points = 0 

114 for group in groups or []: 

115 gid = group.get("group_id") 

116 g_points = group.get("points", 1) 

117 max_points += g_points 

118 g_correct = is_answer_correct( 

119 answers_by_group.get(gid), 

120 (group.get("correctAnswer") or {}).get("answers"), 

121 group.get("type"), 

122 unordered=(group.get("correctAnswer") or {}).get("unordered", False), 

123 ) 

124 g_earned = g_points if g_correct else 0 

125 earned_points += g_earned 

126 group_results[gid] = {"isCorrect": g_correct, "earnedPoints": g_earned, "maxPoints": g_points} 

127 return { 

128 "is_correct": bool(groups) and earned_points == max_points, 

129 "earned_points": earned_points, 

130 "max_points": max_points, 

131 "group_results": group_results, 

132 } 

133 

134 

135def resolve_correct_answer(question_doc: dict): 

136 """What `is_answer_correct`/`score_question` should treat as `correct_answer` 

137 for a question document. 

138 

139 Single-Stimulus and Multi-Part-Question have no flat 

140 `correctAnswer.answers` — each group carries its own — so this returns 

141 the `groups` list itself for those types; every other type keeps 

142 returning `correctAnswer.answers` as before. 

143 """ 

144 if is_multi_group_type(question_doc.get("questionType")): 

145 return question_doc.get("groups", []) 

146 return (question_doc.get("correctAnswer") or {}).get("answers") 

147 

148 

149# Types a person has to read before they can be scored. A written response is graded 

150# here by exact text match, which is right for "name the capital of France" and hopeless 

151# for "explain photosynthesis": the student writes a paragraph, the stored model answer 

152# is a different paragraph, and the match fails. Imported STAAR items are the extreme 

153# case — their stored answer is literally "(See Scoring Guide)", so the only student who 

154# could ever score is one who types that phrase. 

155# 

156# So a written answer the auto-grader cannot CONFIRM is no longer called wrong. It is 

157# held for a person to mark. Silently scoring it zero is the failure this replaces: it 

158# looks like a graded question and it is really an unanswerable one. 

159MANUALLY_MARKED_TYPES = ("free-response",) 

160 

161 

162def needs_manual_marking(question_type, is_correct) -> bool: 

163 """Whether this answer must be read by a person before it can carry a score. 

164 

165 Only ever true when the auto-grader did NOT already confirm the answer. An exact 

166 match still scores itself, so a short-answer question that has always auto-graded 

167 correctly keeps doing so and creates no marking work. 

168 """ 

169 return question_type in MANUALLY_MARKED_TYPES and not is_correct 

170 

171 

172def score_question( 

173 student_answer, correct_answer, question_type, max_points, graph_fingerprint=None, unordered=False 

174): 

175 """Score one answer, returning `(is_correct, earned_points, group_results)`. 

176 

177 `group_results` is `None` for every type except Single-Stimulus and 

178 Multi-Part-Question, which get partial credit — every other type keeps 

179 the existing binary max_points-or-0 behavior unchanged. `correct_answer` 

180 should come from `resolve_correct_answer(question_doc)` so it carries 

181 `groups` for those types. `unordered` only affects the 

182 _POSITION_SENSITIVE_TYPES (Drop-down-Menu et al) — see `is_answer_correct` 

183 — and should come from the question's `correctAnswer.unordered`; every 

184 Single-Stimulus/Multi-Part-Question group reads its own instead (see 

185 `grade_single_stimulus_answer`), so `unordered` here is ignored for those. 

186 """ 

187 if is_multi_group_type(question_type): 

188 result = grade_single_stimulus_answer(student_answer, correct_answer) 

189 return result["is_correct"], result["earned_points"], result["group_results"] 

190 is_correct = is_answer_correct( 

191 student_answer, correct_answer, question_type, graph_fingerprint=graph_fingerprint, unordered=unordered 

192 ) 

193 return is_correct, (max_points if is_correct else 0), None 

194 

195 

196def _build_id_answer_map(entries): 

197 """ 

198 Reduce a list of {"id": ..., "answer": ...} dicts to an id -> answer map. 

199 

200 Returns None (never raises) when the ids can't unambiguously identify a 

201 position: a missing id, a duplicate id, or an unhashable id. Building a 

202 map instead of sorting the raw dicts means the comparison never needs a 

203 total order over `id` — the previous `sorted(entries, key=lambda x: 

204 x["id"])` raised TypeError whenever an id was missing/None (the 

205 question-bank model's Answer.id is Optional[int], so nothing guarantees 

206 every entry has one) or mixed types across entries, which crashed the 

207 entire submission instead of just failing to grade one question. 

208 """ 

209 result = {} 

210 for entry in entries: 

211 try: 

212 entry_id = entry["id"] 

213 if entry_id in result: 

214 return None 

215 result[entry_id] = entry.get("answer") 

216 except TypeError: 

217 return None 

218 return result 

219 

220 

221def is_answer_correct( 

222 student_answer, correct_answer, question_type=None, graph_fingerprint=None, unordered=False 

223) -> bool: 

224 """ 

225 Compare student_answer and correct_answer to determine correctness. 

226 Handles cases for: 

227 - Graph-Multiple-Select (interactive-dot positions + active flags only) 

228 - Graph questions (canonical fingerprint + geometric checkGraphData) 

229 - strings 

230 - lists of strings 

231 - lists of objects with 'id' and 'answer' 

232 

233 free-response is graded separately (see normalize_free_response_answer) 

234 rather than falling into the plain string-equality branch below — 

235 exact-match on raw HTML/LaTeX is essentially never true even for a 

236 genuinely matching answer. A missing/blank correct answer (common for 

237 free-response questions that are graded manually, not auto-graded) 

238 never auto-scores as correct, even against a blank student answer. 

239 

240 `unordered` (from the question's `correctAnswer.unordered` — see 

241 CorrectAnswer in models/question_bank.py) only affects 

242 _POSITION_SENSITIVE_TYPES. Ignored for every other type/shape. 

243 """ 

244 if is_interactive_dots_question_type(question_type): 

245 return grade_interactive_dots_answer(student_answer, correct_answer) 

246 

247 if is_graph_question_type(question_type): 

248 return grade_graph_answer( 

249 student_answer, 

250 correct_answer, 

251 stored_fingerprint=graph_fingerprint, 

252 ) 

253 

254 if is_multi_group_type(question_type): 

255 # `correct_answer` is repurposed to mean "the `groups` list" for 

256 # these types only — see resolve_correct_answer(). Keeps this 

257 # function's signature unchanged for every other call site. 

258 return grade_single_stimulus_answer(student_answer, correct_answer)["is_correct"] 

259 

260 if question_type == "free-response": 

261 normalized_correct = normalize_free_response_answer(correct_answer) 

262 if not normalized_correct: 

263 return False 

264 return normalize_free_response_answer(student_answer) == normalized_correct 

265 

266 normalized_type = _normalize_question_type(question_type) 

267 

268 if isinstance(correct_answer, str): 

269 is_correct = student_answer == correct_answer 

270 

271 elif isinstance(correct_answer, list): 

272 # Check if it's a list of strings 

273 if all(isinstance(ans, str) for ans in correct_answer): 

274 # A list of correct-answer strings means two different things 

275 # depending on the type, and `question_type` — not the shape of 

276 # `student_answer` — is what disambiguates them: 

277 # - Checkbox: student must select every one of them (exact set 

278 # match) — the student's answer is inherently a list here. 

279 # - Multiple-choice (and everything else): the student picks 

280 # ONE option, and any one of several teacher-marked answers 

281 # counts as correct — membership, not set equality. Bug fix: 

282 # this used to branch on `isinstance(student_answer, list)` 

283 # alone, so a multiple-choice answer that happened to arrive 

284 # as a single-element list (rather than a bare string) fell 

285 # into the exact-set-match branch instead and was marked 

286 # wrong even though its one selection was a valid answer. 

287 if normalized_type == "checkbox": 

288 is_correct = isinstance(student_answer, list) and set(student_answer) == set(correct_answer) 

289 elif isinstance(student_answer, list): 

290 is_correct = len(student_answer) == 1 and student_answer[0] in correct_answer 

291 else: 

292 is_correct = student_answer in correct_answer 

293 

294 # Check if it's a list of dicts with 'id' and 'answer' 

295 elif all(isinstance(ans, dict) and "id" in ans and "answer" in ans for ans in correct_answer): 

296 is_position_sensitive = normalized_type in _POSITION_SENSITIVE_TYPES 

297 if unordered and is_position_sensitive and isinstance(student_answer, list) and student_answer: 

298 # The teacher opted this question into UNORDERED grading (see 

299 # CorrectAnswer.unordered) — a correct answer counts in ANY 

300 # blank, not only the one it was authored for, so grade as a 

301 # multiset of texts rather than id-to-id / position-to-position. 

302 # Takes priority over both branches below, regardless of 

303 # whether the student's answer happens to carry ids. 

304 correct_texts = [ans.get("answer") for ans in correct_answer] 

305 student_texts = [a.get("answer") if isinstance(a, dict) else a for a in student_answer] 

306 if normalized_type in ("drop-down", "drop-down-menu"): 

307 # Same HTML-vs-plain-text normalization as the ordered 

308 # id-map path below — see that branch's comment. 

309 correct_texts = [normalize_free_response_answer(t) for t in correct_texts] 

310 student_texts = [normalize_free_response_answer(t) for t in student_texts] 

311 is_correct = sorted(map(str, student_texts)) == sorted(map(str, correct_texts)) 

312 # Only take the id-keyed dict-vs-dict path when the student 

313 # answer is ALSO a list of {id, ...} dicts. A list of dicts 

314 # WITHOUT an "id" key here raised KeyError; fall through to the 

315 # text comparison below instead. 

316 elif ( 

317 isinstance(student_answer, list) 

318 and student_answer 

319 and all(isinstance(ans, dict) and "id" in ans for ans in student_answer) 

320 ): 

321 # Compare as an id -> answer mapping, not sorted raw dicts. 

322 # This is position-sensitive (each id is a distinct 

323 # blank/drop-target, so a correct answer assigned to the 

324 # WRONG id still fails) and never raises — see 

325 # _build_id_answer_map for why sorting was unsafe here. 

326 correct_map = _build_id_answer_map(correct_answer) 

327 student_map = _build_id_answer_map(student_answer) 

328 if ( 

329 normalized_type in ("drop-down", "drop-down-menu") 

330 and correct_map is not None 

331 and student_map is not None 

332 ): 

333 # A Single-Stimulus "drop-down" group's correct answer 

334 # (and the standalone "drop-down-menu" question type's, 

335 # same shape — DropdownMenuV2.jsx is shared by both 

336 # editors) is authored as PLAIN TEXT (DropdownMenuV2.jsx's 

337 # onClick does tinyMCEtoString(choice.text)), but a 

338 # student's chosen option comes back as that option's raw 

339 # rich-text HTML (GroupDropdownAnswer.jsx / DropDownMenu.jsx's 

340 # <select> reflects the <option value> verbatim — e.g. 

341 # "<p>100C</p>" vs "100C"). Comparing the two verbatim 

342 # would fail every correct answer, so normalize both sides 

343 # the same way normalize_free_response_answer already does 

344 # for free-response before comparing. 

345 correct_map = {k: normalize_free_response_answer(v) for k, v in correct_map.items()} 

346 student_map = {k: normalize_free_response_answer(v) for k, v in student_map.items()} 

347 is_correct = correct_map is not None and correct_map == student_map 

348 elif is_position_sensitive and isinstance(student_answer, list) and student_answer: 

349 # No ids on the student's side for a position-sensitive type 

350 # (Drop-down-Menu / Drag-and-Drop), but the student DID submit 

351 # a list — array order is itself the position here (ids are 

352 # assigned sequentially by position at authoring time, and 

353 # never reordered independently of it — see 

354 # DropdownMenuV2.jsx / the CSV importer), so zip the two 

355 # lists by index instead of comparing them as an unordered 

356 # bag of text. Comparing as a multiset (like the checkbox 

357 # path below) would credit two correct values swapped 

358 # between different blanks as fully correct; zipping by 

359 # position catches that mismatch while still correctly 

360 # scoring the common "answered in order, just without ids" 

361 # shape. 

362 if len(student_answer) != len(correct_answer): 

363 is_correct = False 

364 else: 

365 correct_texts_by_pos = [ans.get("answer") for ans in correct_answer] 

366 student_texts_by_pos = [ 

367 a.get("answer") if isinstance(a, dict) else a 

368 for a in student_answer 

369 ] 

370 is_correct = student_texts_by_pos == correct_texts_by_pos 

371 else: 

372 # Global / staff questions store correctAnswer as a list of 

373 # {id, answer} dicts, but the student submits the plain answer 

374 # TEXT — a string (multiple-choice) or a list of strings 

375 # (checkbox), where order genuinely doesn't matter — or a 

376 # position-sensitive type answered as a single bare scalar 

377 # rather than a list at all, which has no position to get 

378 # wrong. The previous `else: False` scored every dict-shaped 

379 # submission wrong (grade 0 in the gradebook) even when the 

380 # answer matched. Compare against the dicts' `answer` values 

381 # (exact match — HTML/LaTeX answer strings are submitted 

382 # verbatim). 

383 correct_texts = [ans.get("answer") for ans in correct_answer] 

384 if isinstance(student_answer, list): 

385 # Tolerate dict elements (an id-less list from a non 

386 # position-sensitive type) and mixed lists — extract each 

387 # `answer` and compare as strings so `sorted` never 

388 # raises TypeError on uncomparable types. 

389 student_texts = [ 

390 a.get("answer") if isinstance(a, dict) else a 

391 for a in student_answer 

392 ] 

393 is_correct = sorted(map(str, student_texts)) == sorted(map(str, correct_texts)) 

394 else: 

395 is_correct = student_answer in correct_texts 

396 

397 else: 

398 is_correct = False 

399 else: 

400 is_correct = False 

401 

402 return is_correct