Coverage for server / models / assignment_submit.py: 95%

44 statements  

« prev     ^ index     » next       coverage.py v7.13.4, created at 2026-10-04 09:33 +0000

1import json 

2 

3from pydantic import BaseModel, Field, field_validator, model_validator 

4 

5from server.utilities.html_sanitizer import sanitize_answer_value 

6from typing import Union, List, Dict, Any, Optional 

7 

8# Defensive bounds on the student-controlled answer payload (EI-2906). The answer 

9# is stored verbatim into the submission, so cap its size so a single submit 

10# can't bloat the document / grading work. 

11_MAX_ANSWER_STR = 20000 # generous for rich-text / LaTeX answers 

12_MAX_ANSWER_ITEMS = 500 # per-question answer entries (drag/checkbox blanks) 

13_MAX_ANSWERS = 500 # questions per submission 

14_MAX_ANSWER_BYTES = 100000 # total serialized size of one answer (covers dict/nested) 

15# Total serialized size across ALL answers — keep the WHOLE submission well under 

16# MongoDB's 16MB BSON limit so a max-per-answer × max-answers payload can't be 

17# accepted only to fail the write (EI-3271 hardening). 

18_MAX_TOTAL_ANSWER_BYTES = 4000000 

19 

20 

21class AnswerItem(BaseModel): 

22 questionId: str = Field(..., max_length=64) 

23 studentAnswer: Union[str, List[Union[str, Dict[str, Any]]], Dict[str, Any]] = "" 

24 isFlagged: Optional[bool] = None 

25 

26 @field_validator("studentAnswer") 

27 @classmethod 

28 def _bound_answer_size(cls, v): 

29 if isinstance(v, str) and len(v) > _MAX_ANSWER_STR: 

30 raise ValueError("studentAnswer is too large") 

31 if isinstance(v, list) and len(v) > _MAX_ANSWER_ITEMS: 

32 raise ValueError("studentAnswer has too many items") 

33 # Per-answer serialized-size cap — covers the dict branch and nested 

34 # dict/str content the str/list length caps above don't reach 

35 # (EI-3271 hardening). json.dumps(default=str) never raises here. 

36 if len(json.dumps(v, default=str)) > _MAX_ANSWER_BYTES: 

37 raise ValueError("studentAnswer payload is too large") 

38 return v 

39 

40 @field_validator("studentAnswer") 

41 @classmethod 

42 def _neutralise_markup(cls, v): 

43 """Sanitise the answer's markup before it is stored (EI-T717 / EI-T721). 

44 

45 Added by Allan Ninal — 2026-09-23 

46 WHAT: run every string inside `studentAnswer` through 

47 `sanitize_rich_text` (nh3 allow-list), on the MODEL so both 

48 /answers/save and /answers/submit are covered by one rule. 

49 WHY: the answer was stored and echoed back byte-for-byte, so a 

50 `<script>` a student typed came straight back out — what EI-T717 

51 and EI-T721 recorded. 

52 

53 SANITISE, NOT STRIP. A student answer is rich HTML from 

54 TinyMCEAnswerEditor, and a typed formula is stored as 

55 `<span class="mfe-formula" data-latex="...">`. `strip_html` would 

56 destroy both the formatting and the maths; 

57 `sanitize_rich_text` keeps the allow-listed markup and drops 

58 `<script>` (tag AND content), `on*` handlers and javascript: URIs. It is 

59 already the house tool for teacher question content. 

60 

61 GRADING IS UNAFFECTED: `normalize_free_response_answer` collapses each 

62 formula span to its `data-latex` before comparing, and the allow-list 

63 preserves `span`, `class` and `data-*` — pinned by 

64 tests/unit/models/test_answer_sanitisation.py::TestGradingDoesNotMove. 

65 

66 ORDER: this runs AFTER `_bound_answer_size`, deliberately — the size gate 

67 is cheap and rejects an abusive payload before nh3 is asked to parse it. 

68 That is the opposite of the ContactPerson ordering (PR #326), where 

69 stripping had to come first because markup was NOT legitimate there and 

70 could otherwise be smuggled inside the length budget. Here markup is 

71 legitimate, so there is no budget to smuggle into. 

72 """ 

73 return sanitize_answer_value(v) 

74 

75 

76class AssignmentSubmitRequest(BaseModel): 

77 # Modified by Allan Ninal — 2026-09-23 (EI-T367 save / EI-T371 submit) 

78 # WHAT: added min_length=1 — the bound had an upper limit and no lower one. 

79 # WHY: `{"answers": []}` validated and reached the service, so a save 

80 # reported success and a submit GRADED THE EMPTY ATTEMPT 0 and recorded 

81 # it complete. One model backs both routes, so one bound closes both. 

82 # This rejects an empty ARRAY, not a blank answer: a student who 

83 # answered nothing still sends one entry per question with 

84 # studentAnswer "", which stays valid and still scores 0. The SPA builds 

85 # the payload that way (one entry per question), so an empty array can 

86 # only come from a hand-crafted request. 

87 answers: List[AnswerItem] = Field(..., min_length=1, max_length=_MAX_ANSWERS) 

88 

89 @model_validator(mode="after") 

90 def _bound_total_payload(self): 

91 # Cap the TOTAL answer payload so the stored submission stays well under 

92 # the 16MB BSON limit (per-answer × max-answers would otherwise allow ~50MB). 

93 total = sum(len(json.dumps(a.studentAnswer, default=str)) for a in self.answers) 

94 if total > _MAX_TOTAL_ANSWER_BYTES: 

95 raise ValueError("answers payload is too large") 

96 return self 

97 

98 

99class ViolationReportRequest(BaseModel): 

100 """One browser-lockdown violation report (see the student assignment 

101 violations endpoint). `type` is informational only — stored for debugging 

102 (last_violation_type/last_violation_at on the submission), not scored or 

103 otherwise interpreted server-side.""" 

104 

105 type: Optional[str] = Field(default=None, max_length=64) 

106 

107 @field_validator("type") 

108 @classmethod 

109 def _blank_is_no_type(cls, value: Optional[str]) -> Optional[str]: 

110 """A blank `type` is stored as None, not as "" or " ". 

111 

112 Added by Allan Ninal — 2026-09-24 (EI-T374). 

113 WHAT: strip the value, and map an empty or whitespace-only one to None. 

114 WHY: `report_violation` writes this straight to `last_violation_type` on 

115 the submission — the RAW client value, unlike the per-type breakdown 

116 key, which is bucketed through `_normalize_violation_type` against an 

117 allow-list. So "" and " " were being stored verbatim in a field 

118 whose entire purpose is telling a human what the client sent. None 

119 says "the client sent nothing" honestly; " " only looks like data. 

120 

121 Note what this deliberately does NOT do: it does not reject the report. 

122 A browser-lockdown violation is a proctoring signal, and refusing one over 

123 a missing label would drop the signal entirely — the violation would go 

124 uncounted and the student would keep working. The count is unaffected by a 

125 blank type either way (`_normalize_violation_type` buckets it as 

126 "unknown"), so there is nothing to gain by failing closed and an integrity 

127 signal to lose. See EI-T374 / EI-T722 for the recorded expectations this 

128 departs from, and the reasoning posted there. 

129 """ 

130 if value is None: 

131 return None 

132 stripped = value.strip() 

133 return stripped or None