Coverage for server / models / assignment_submit.py: 95%
44 statements
« prev ^ index » next coverage.py v7.13.4, created at 2026-10-04 09:33 +0000
« prev ^ index » next coverage.py v7.13.4, created at 2026-10-04 09:33 +0000
1import json
3from pydantic import BaseModel, Field, field_validator, model_validator
5from server.utilities.html_sanitizer import sanitize_answer_value
6from typing import Union, List, Dict, Any, Optional
8# Defensive bounds on the student-controlled answer payload (EI-2906). The answer
9# is stored verbatim into the submission, so cap its size so a single submit
10# can't bloat the document / grading work.
11_MAX_ANSWER_STR = 20000 # generous for rich-text / LaTeX answers
12_MAX_ANSWER_ITEMS = 500 # per-question answer entries (drag/checkbox blanks)
13_MAX_ANSWERS = 500 # questions per submission
14_MAX_ANSWER_BYTES = 100000 # total serialized size of one answer (covers dict/nested)
15# Total serialized size across ALL answers — keep the WHOLE submission well under
16# MongoDB's 16MB BSON limit so a max-per-answer × max-answers payload can't be
17# accepted only to fail the write (EI-3271 hardening).
18_MAX_TOTAL_ANSWER_BYTES = 4000000
21class AnswerItem(BaseModel):
22 questionId: str = Field(..., max_length=64)
23 studentAnswer: Union[str, List[Union[str, Dict[str, Any]]], Dict[str, Any]] = ""
24 isFlagged: Optional[bool] = None
26 @field_validator("studentAnswer")
27 @classmethod
28 def _bound_answer_size(cls, v):
29 if isinstance(v, str) and len(v) > _MAX_ANSWER_STR:
30 raise ValueError("studentAnswer is too large")
31 if isinstance(v, list) and len(v) > _MAX_ANSWER_ITEMS:
32 raise ValueError("studentAnswer has too many items")
33 # Per-answer serialized-size cap — covers the dict branch and nested
34 # dict/str content the str/list length caps above don't reach
35 # (EI-3271 hardening). json.dumps(default=str) never raises here.
36 if len(json.dumps(v, default=str)) > _MAX_ANSWER_BYTES:
37 raise ValueError("studentAnswer payload is too large")
38 return v
40 @field_validator("studentAnswer")
41 @classmethod
42 def _neutralise_markup(cls, v):
43 """Sanitise the answer's markup before it is stored (EI-T717 / EI-T721).
45 Added by Allan Ninal — 2026-09-23
46 WHAT: run every string inside `studentAnswer` through
47 `sanitize_rich_text` (nh3 allow-list), on the MODEL so both
48 /answers/save and /answers/submit are covered by one rule.
49 WHY: the answer was stored and echoed back byte-for-byte, so a
50 `<script>` a student typed came straight back out — what EI-T717
51 and EI-T721 recorded.
53 SANITISE, NOT STRIP. A student answer is rich HTML from
54 TinyMCEAnswerEditor, and a typed formula is stored as
55 `<span class="mfe-formula" data-latex="...">`. `strip_html` would
56 destroy both the formatting and the maths;
57 `sanitize_rich_text` keeps the allow-listed markup and drops
58 `<script>` (tag AND content), `on*` handlers and javascript: URIs. It is
59 already the house tool for teacher question content.
61 GRADING IS UNAFFECTED: `normalize_free_response_answer` collapses each
62 formula span to its `data-latex` before comparing, and the allow-list
63 preserves `span`, `class` and `data-*` — pinned by
64 tests/unit/models/test_answer_sanitisation.py::TestGradingDoesNotMove.
66 ORDER: this runs AFTER `_bound_answer_size`, deliberately — the size gate
67 is cheap and rejects an abusive payload before nh3 is asked to parse it.
68 That is the opposite of the ContactPerson ordering (PR #326), where
69 stripping had to come first because markup was NOT legitimate there and
70 could otherwise be smuggled inside the length budget. Here markup is
71 legitimate, so there is no budget to smuggle into.
72 """
73 return sanitize_answer_value(v)
76class AssignmentSubmitRequest(BaseModel):
77 # Modified by Allan Ninal — 2026-09-23 (EI-T367 save / EI-T371 submit)
78 # WHAT: added min_length=1 — the bound had an upper limit and no lower one.
79 # WHY: `{"answers": []}` validated and reached the service, so a save
80 # reported success and a submit GRADED THE EMPTY ATTEMPT 0 and recorded
81 # it complete. One model backs both routes, so one bound closes both.
82 # This rejects an empty ARRAY, not a blank answer: a student who
83 # answered nothing still sends one entry per question with
84 # studentAnswer "", which stays valid and still scores 0. The SPA builds
85 # the payload that way (one entry per question), so an empty array can
86 # only come from a hand-crafted request.
87 answers: List[AnswerItem] = Field(..., min_length=1, max_length=_MAX_ANSWERS)
89 @model_validator(mode="after")
90 def _bound_total_payload(self):
91 # Cap the TOTAL answer payload so the stored submission stays well under
92 # the 16MB BSON limit (per-answer × max-answers would otherwise allow ~50MB).
93 total = sum(len(json.dumps(a.studentAnswer, default=str)) for a in self.answers)
94 if total > _MAX_TOTAL_ANSWER_BYTES:
95 raise ValueError("answers payload is too large")
96 return self
99class ViolationReportRequest(BaseModel):
100 """One browser-lockdown violation report (see the student assignment
101 violations endpoint). `type` is informational only — stored for debugging
102 (last_violation_type/last_violation_at on the submission), not scored or
103 otherwise interpreted server-side."""
105 type: Optional[str] = Field(default=None, max_length=64)
107 @field_validator("type")
108 @classmethod
109 def _blank_is_no_type(cls, value: Optional[str]) -> Optional[str]:
110 """A blank `type` is stored as None, not as "" or " ".
112 Added by Allan Ninal — 2026-09-24 (EI-T374).
113 WHAT: strip the value, and map an empty or whitespace-only one to None.
114 WHY: `report_violation` writes this straight to `last_violation_type` on
115 the submission — the RAW client value, unlike the per-type breakdown
116 key, which is bucketed through `_normalize_violation_type` against an
117 allow-list. So "" and " " were being stored verbatim in a field
118 whose entire purpose is telling a human what the client sent. None
119 says "the client sent nothing" honestly; " " only looks like data.
121 Note what this deliberately does NOT do: it does not reject the report.
122 A browser-lockdown violation is a proctoring signal, and refusing one over
123 a missing label would drop the signal entirely — the violation would go
124 uncounted and the student would keep working. The count is unaffected by a
125 blank type either way (`_normalize_violation_type` buckets it as
126 "unknown"), so there is nothing to gain by failing closed and an integrity
127 signal to lose. See EI-T374 / EI-T722 for the recorded expectations this
128 departs from, and the reasoning posted there.
129 """
130 if value is None:
131 return None
132 stripped = value.strip()
133 return stripped or None