Coverage for server / services / common / answer_checking.py: 96%
111 statements
« prev ^ index » next coverage.py v7.13.4, created at 2026-10-04 09:33 +0000
« prev ^ index » next coverage.py v7.13.4, created at 2026-10-04 09:33 +0000
1"""Canonical correct-answer checking, shared across every question type.
3Single source of truth for "is this submitted answer correct" — the student
4assignment-submit/scoring path and teacher-side reporting (Item Analysis,
5live submission view) both call `is_answer_correct` so a submission is never
6graded one way by the gradebook and reported a different way in analytics.
7"""
8import html
9import re
11from server.utilities.graph_data_checker import (
12 grade_graph_answer,
13 grade_interactive_dots_answer,
14 is_graph_question_type,
15 is_interactive_dots_question_type,
16)
19# Matches a math-formula span as inserted by the TinyMCE math-formula plugin:
20# <span class="mfe-formula" data-latex="...">...rendered markup...</span>.
21# Used to canonicalize a free-response answer for grading — see
22# normalize_free_response_answer.
23MFE_FORMULA_RE = re.compile(
24 r'<span[^>]*class="mfe-formula"[^>]*data-latex="([^"]*)"[^>]*>.*?</span>',
25 re.DOTALL,
26)
29def normalize_free_response_answer(value) -> str:
30 """
31 Canonicalizes a free-response answer for comparison.
33 The stored value is TinyMCE HTML (the answer/model-answer editor —
34 see FreeResponse.jsx/TinyMCEAnswerEditor.jsx in the client — wraps a
35 typed math expression in `<span class="mfe-formula" data-latex="...">
36 ...rendered markup...</span>`), or, for older submissions made before
37 that editor existed, a plain LaTeX string typed into a math-only
38 field. A byte-for-byte comparison of the raw HTML is not a
39 meaningful equality check the way it is for a chosen multiple-choice
40 option (which round-trips the exact stored choice string verbatim):
41 two independently-typed answers containing "the same" formula do not
42 produce identical rendered markup, and a plain LaTeX string can never
43 equal an HTML string at all.
45 Each math-formula span is collapsed to its `data-latex` source
46 (stable across renders/typing sessions) instead of its rendered
47 markup, remaining HTML tags are stripped, entities are decoded, and
48 whitespace is collapsed — so paragraph/line-break/formatting
49 differences that don't change the visible answer don't cause a
50 false negative.
51 """
52 if isinstance(value, dict):
53 value = value.get("answer", "")
54 if not isinstance(value, str):
55 value = "" if value is None else str(value)
57 value = MFE_FORMULA_RE.sub(lambda m: m.group(1), value)
58 value = re.sub(r"<[^>]+>", " ", value)
59 value = html.unescape(value)
60 return re.sub(r"\s+", " ", value).strip()
63# Question types where each {id, answer} entry is a distinct blank/position
64# (a dropdown, a drop target) rather than an independently-correct option in
65# an unordered set (checkbox). Grading these must compare answers id-to-id,
66# not as a bag of texts — see _build_id_answer_map. "drop-down" is the
67# Single-Stimulus "drop-down" group type (group.get("type")) — distinct
68# from the standalone "drop-down-menu" question type, but the same
69# id-per-blank shape (see SingleStimulusEditor.jsx / DropdownMenuV2.jsx
70# on the client), so it gets the same position-sensitive fallback.
71# "grid-question" reuses this same id-to-id comparison unchanged: each row
72# is a "blank" (id = row id, answer = the chosen column's text) — see
73# docs/features/grid-question.md.
74_POSITION_SENSITIVE_TYPES = frozenset({"drop-down-menu", "drag-and-drop", "drop-down", "grid-question"})
77def _normalize_question_type(question_type) -> str | None:
78 if question_type is None:
79 return None
80 return str(question_type).strip().lower().replace("_", "-")
83_GROUP_GRADED_TYPES = frozenset({"single-stimulus", "multi-part-question"})
86def is_multi_group_type(question_type) -> bool:
87 """Whether `question_type` is graded group-by-group (Single-Stimulus,
88 Multi-Part-Question) rather than as one flat answer. Both share the same
89 `groups` shape and grading — Multi-Part-Question's only difference is an
90 authoring-time-only per-group `questionText` field this module never
91 reads."""
92 return _normalize_question_type(question_type) in _GROUP_GRADED_TYPES
95def grade_single_stimulus_answer(student_answer, groups) -> dict:
96 """Grade a Single-Stimulus answer — each group scored independently.
98 Args:
99 student_answer: the combined answer for the question, shaped
100 ``{group_id: <that group's answer, same shape its own `type`
101 normally submits>}`` (or anything else / missing for unanswered).
102 groups: the question document's `groups` list — each carries its own
103 `type`/`choices`/`correctAnswer`/`points`.
105 Returns a dict with `is_correct` (all groups correct), `earned_points`,
106 `max_points`, and `group_results` (per-group `{isCorrect, earnedPoints,
107 maxPoints}`, keyed by group_id) — a correct answer in one group never
108 affects another's score.
109 """
110 answers_by_group = student_answer if isinstance(student_answer, dict) else {}
111 group_results: dict = {}
112 earned_points = 0
113 max_points = 0
114 for group in groups or []:
115 gid = group.get("group_id")
116 g_points = group.get("points", 1)
117 max_points += g_points
118 g_correct = is_answer_correct(
119 answers_by_group.get(gid),
120 (group.get("correctAnswer") or {}).get("answers"),
121 group.get("type"),
122 unordered=(group.get("correctAnswer") or {}).get("unordered", False),
123 )
124 g_earned = g_points if g_correct else 0
125 earned_points += g_earned
126 group_results[gid] = {"isCorrect": g_correct, "earnedPoints": g_earned, "maxPoints": g_points}
127 return {
128 "is_correct": bool(groups) and earned_points == max_points,
129 "earned_points": earned_points,
130 "max_points": max_points,
131 "group_results": group_results,
132 }
135def resolve_correct_answer(question_doc: dict):
136 """What `is_answer_correct`/`score_question` should treat as `correct_answer`
137 for a question document.
139 Single-Stimulus and Multi-Part-Question have no flat
140 `correctAnswer.answers` — each group carries its own — so this returns
141 the `groups` list itself for those types; every other type keeps
142 returning `correctAnswer.answers` as before.
143 """
144 if is_multi_group_type(question_doc.get("questionType")):
145 return question_doc.get("groups", [])
146 return (question_doc.get("correctAnswer") or {}).get("answers")
149# Types a person has to read before they can be scored. A written response is graded
150# here by exact text match, which is right for "name the capital of France" and hopeless
151# for "explain photosynthesis": the student writes a paragraph, the stored model answer
152# is a different paragraph, and the match fails. Imported STAAR items are the extreme
153# case — their stored answer is literally "(See Scoring Guide)", so the only student who
154# could ever score is one who types that phrase.
155#
156# So a written answer the auto-grader cannot CONFIRM is no longer called wrong. It is
157# held for a person to mark. Silently scoring it zero is the failure this replaces: it
158# looks like a graded question and it is really an unanswerable one.
159MANUALLY_MARKED_TYPES = ("free-response",)
162def needs_manual_marking(question_type, is_correct) -> bool:
163 """Whether this answer must be read by a person before it can carry a score.
165 Only ever true when the auto-grader did NOT already confirm the answer. An exact
166 match still scores itself, so a short-answer question that has always auto-graded
167 correctly keeps doing so and creates no marking work.
168 """
169 return question_type in MANUALLY_MARKED_TYPES and not is_correct
172def score_question(
173 student_answer, correct_answer, question_type, max_points, graph_fingerprint=None, unordered=False
174):
175 """Score one answer, returning `(is_correct, earned_points, group_results)`.
177 `group_results` is `None` for every type except Single-Stimulus and
178 Multi-Part-Question, which get partial credit — every other type keeps
179 the existing binary max_points-or-0 behavior unchanged. `correct_answer`
180 should come from `resolve_correct_answer(question_doc)` so it carries
181 `groups` for those types. `unordered` only affects the
182 _POSITION_SENSITIVE_TYPES (Drop-down-Menu et al) — see `is_answer_correct`
183 — and should come from the question's `correctAnswer.unordered`; every
184 Single-Stimulus/Multi-Part-Question group reads its own instead (see
185 `grade_single_stimulus_answer`), so `unordered` here is ignored for those.
186 """
187 if is_multi_group_type(question_type):
188 result = grade_single_stimulus_answer(student_answer, correct_answer)
189 return result["is_correct"], result["earned_points"], result["group_results"]
190 is_correct = is_answer_correct(
191 student_answer, correct_answer, question_type, graph_fingerprint=graph_fingerprint, unordered=unordered
192 )
193 return is_correct, (max_points if is_correct else 0), None
196def _build_id_answer_map(entries):
197 """
198 Reduce a list of {"id": ..., "answer": ...} dicts to an id -> answer map.
200 Returns None (never raises) when the ids can't unambiguously identify a
201 position: a missing id, a duplicate id, or an unhashable id. Building a
202 map instead of sorting the raw dicts means the comparison never needs a
203 total order over `id` — the previous `sorted(entries, key=lambda x:
204 x["id"])` raised TypeError whenever an id was missing/None (the
205 question-bank model's Answer.id is Optional[int], so nothing guarantees
206 every entry has one) or mixed types across entries, which crashed the
207 entire submission instead of just failing to grade one question.
208 """
209 result = {}
210 for entry in entries:
211 try:
212 entry_id = entry["id"]
213 if entry_id in result:
214 return None
215 result[entry_id] = entry.get("answer")
216 except TypeError:
217 return None
218 return result
221def is_answer_correct(
222 student_answer, correct_answer, question_type=None, graph_fingerprint=None, unordered=False
223) -> bool:
224 """
225 Compare student_answer and correct_answer to determine correctness.
226 Handles cases for:
227 - Graph-Multiple-Select (interactive-dot positions + active flags only)
228 - Graph questions (canonical fingerprint + geometric checkGraphData)
229 - strings
230 - lists of strings
231 - lists of objects with 'id' and 'answer'
233 free-response is graded separately (see normalize_free_response_answer)
234 rather than falling into the plain string-equality branch below —
235 exact-match on raw HTML/LaTeX is essentially never true even for a
236 genuinely matching answer. A missing/blank correct answer (common for
237 free-response questions that are graded manually, not auto-graded)
238 never auto-scores as correct, even against a blank student answer.
240 `unordered` (from the question's `correctAnswer.unordered` — see
241 CorrectAnswer in models/question_bank.py) only affects
242 _POSITION_SENSITIVE_TYPES. Ignored for every other type/shape.
243 """
244 if is_interactive_dots_question_type(question_type):
245 return grade_interactive_dots_answer(student_answer, correct_answer)
247 if is_graph_question_type(question_type):
248 return grade_graph_answer(
249 student_answer,
250 correct_answer,
251 stored_fingerprint=graph_fingerprint,
252 )
254 if is_multi_group_type(question_type):
255 # `correct_answer` is repurposed to mean "the `groups` list" for
256 # these types only — see resolve_correct_answer(). Keeps this
257 # function's signature unchanged for every other call site.
258 return grade_single_stimulus_answer(student_answer, correct_answer)["is_correct"]
260 if question_type == "free-response":
261 normalized_correct = normalize_free_response_answer(correct_answer)
262 if not normalized_correct:
263 return False
264 return normalize_free_response_answer(student_answer) == normalized_correct
266 normalized_type = _normalize_question_type(question_type)
268 if isinstance(correct_answer, str):
269 is_correct = student_answer == correct_answer
271 elif isinstance(correct_answer, list):
272 # Check if it's a list of strings
273 if all(isinstance(ans, str) for ans in correct_answer):
274 # A list of correct-answer strings means two different things
275 # depending on the type, and `question_type` — not the shape of
276 # `student_answer` — is what disambiguates them:
277 # - Checkbox: student must select every one of them (exact set
278 # match) — the student's answer is inherently a list here.
279 # - Multiple-choice (and everything else): the student picks
280 # ONE option, and any one of several teacher-marked answers
281 # counts as correct — membership, not set equality. Bug fix:
282 # this used to branch on `isinstance(student_answer, list)`
283 # alone, so a multiple-choice answer that happened to arrive
284 # as a single-element list (rather than a bare string) fell
285 # into the exact-set-match branch instead and was marked
286 # wrong even though its one selection was a valid answer.
287 if normalized_type == "checkbox":
288 is_correct = isinstance(student_answer, list) and set(student_answer) == set(correct_answer)
289 elif isinstance(student_answer, list):
290 is_correct = len(student_answer) == 1 and student_answer[0] in correct_answer
291 else:
292 is_correct = student_answer in correct_answer
294 # Check if it's a list of dicts with 'id' and 'answer'
295 elif all(isinstance(ans, dict) and "id" in ans and "answer" in ans for ans in correct_answer):
296 is_position_sensitive = normalized_type in _POSITION_SENSITIVE_TYPES
297 if unordered and is_position_sensitive and isinstance(student_answer, list) and student_answer:
298 # The teacher opted this question into UNORDERED grading (see
299 # CorrectAnswer.unordered) — a correct answer counts in ANY
300 # blank, not only the one it was authored for, so grade as a
301 # multiset of texts rather than id-to-id / position-to-position.
302 # Takes priority over both branches below, regardless of
303 # whether the student's answer happens to carry ids.
304 correct_texts = [ans.get("answer") for ans in correct_answer]
305 student_texts = [a.get("answer") if isinstance(a, dict) else a for a in student_answer]
306 if normalized_type in ("drop-down", "drop-down-menu"):
307 # Same HTML-vs-plain-text normalization as the ordered
308 # id-map path below — see that branch's comment.
309 correct_texts = [normalize_free_response_answer(t) for t in correct_texts]
310 student_texts = [normalize_free_response_answer(t) for t in student_texts]
311 is_correct = sorted(map(str, student_texts)) == sorted(map(str, correct_texts))
312 # Only take the id-keyed dict-vs-dict path when the student
313 # answer is ALSO a list of {id, ...} dicts. A list of dicts
314 # WITHOUT an "id" key here raised KeyError; fall through to the
315 # text comparison below instead.
316 elif (
317 isinstance(student_answer, list)
318 and student_answer
319 and all(isinstance(ans, dict) and "id" in ans for ans in student_answer)
320 ):
321 # Compare as an id -> answer mapping, not sorted raw dicts.
322 # This is position-sensitive (each id is a distinct
323 # blank/drop-target, so a correct answer assigned to the
324 # WRONG id still fails) and never raises — see
325 # _build_id_answer_map for why sorting was unsafe here.
326 correct_map = _build_id_answer_map(correct_answer)
327 student_map = _build_id_answer_map(student_answer)
328 if (
329 normalized_type in ("drop-down", "drop-down-menu")
330 and correct_map is not None
331 and student_map is not None
332 ):
333 # A Single-Stimulus "drop-down" group's correct answer
334 # (and the standalone "drop-down-menu" question type's,
335 # same shape — DropdownMenuV2.jsx is shared by both
336 # editors) is authored as PLAIN TEXT (DropdownMenuV2.jsx's
337 # onClick does tinyMCEtoString(choice.text)), but a
338 # student's chosen option comes back as that option's raw
339 # rich-text HTML (GroupDropdownAnswer.jsx / DropDownMenu.jsx's
340 # <select> reflects the <option value> verbatim — e.g.
341 # "<p>100C</p>" vs "100C"). Comparing the two verbatim
342 # would fail every correct answer, so normalize both sides
343 # the same way normalize_free_response_answer already does
344 # for free-response before comparing.
345 correct_map = {k: normalize_free_response_answer(v) for k, v in correct_map.items()}
346 student_map = {k: normalize_free_response_answer(v) for k, v in student_map.items()}
347 is_correct = correct_map is not None and correct_map == student_map
348 elif is_position_sensitive and isinstance(student_answer, list) and student_answer:
349 # No ids on the student's side for a position-sensitive type
350 # (Drop-down-Menu / Drag-and-Drop), but the student DID submit
351 # a list — array order is itself the position here (ids are
352 # assigned sequentially by position at authoring time, and
353 # never reordered independently of it — see
354 # DropdownMenuV2.jsx / the CSV importer), so zip the two
355 # lists by index instead of comparing them as an unordered
356 # bag of text. Comparing as a multiset (like the checkbox
357 # path below) would credit two correct values swapped
358 # between different blanks as fully correct; zipping by
359 # position catches that mismatch while still correctly
360 # scoring the common "answered in order, just without ids"
361 # shape.
362 if len(student_answer) != len(correct_answer):
363 is_correct = False
364 else:
365 correct_texts_by_pos = [ans.get("answer") for ans in correct_answer]
366 student_texts_by_pos = [
367 a.get("answer") if isinstance(a, dict) else a
368 for a in student_answer
369 ]
370 is_correct = student_texts_by_pos == correct_texts_by_pos
371 else:
372 # Global / staff questions store correctAnswer as a list of
373 # {id, answer} dicts, but the student submits the plain answer
374 # TEXT — a string (multiple-choice) or a list of strings
375 # (checkbox), where order genuinely doesn't matter — or a
376 # position-sensitive type answered as a single bare scalar
377 # rather than a list at all, which has no position to get
378 # wrong. The previous `else: False` scored every dict-shaped
379 # submission wrong (grade 0 in the gradebook) even when the
380 # answer matched. Compare against the dicts' `answer` values
381 # (exact match — HTML/LaTeX answer strings are submitted
382 # verbatim).
383 correct_texts = [ans.get("answer") for ans in correct_answer]
384 if isinstance(student_answer, list):
385 # Tolerate dict elements (an id-less list from a non
386 # position-sensitive type) and mixed lists — extract each
387 # `answer` and compare as strings so `sorted` never
388 # raises TypeError on uncomparable types.
389 student_texts = [
390 a.get("answer") if isinstance(a, dict) else a
391 for a in student_answer
392 ]
393 is_correct = sorted(map(str, student_texts)) == sorted(map(str, correct_texts))
394 else:
395 is_correct = student_answer in correct_texts
397 else:
398 is_correct = False
399 else:
400 is_correct = False
402 return is_correct