Coverage for server / services / teacher / teacher_question.py: 88%

737 statements  

« prev     ^ index     » next       coverage.py v7.13.4, created at 2026-10-04 09:33 +0000

1import copy 

2import json 

3import logging 

4import math 

5import os 

6import uuid 

7from typing import Annotated, Type, Dict, Any 

8from bson import ObjectId 

9from fastapi import ( 

10 Body, 

11 Depends, 

12 File, 

13 HTTPException, 

14 Query, 

15 Request, 

16 UploadFile, 

17 status, 

18) 

19from fastapi.openapi.models import Example 

20import filetype 

21from pydantic import ValidationError 

22from pymongo import ReturnDocument 

23from server.connection.database import db, staff_admin_db 

24from server.utilities.answer_match import first_unmatched_answer, normalize_answer_text 

25from server.utilities.html_sanitizer import sanitize_rich_text 

26from server.utilities.user_id_helper import to_user_id 

27from server.connection.storage_bucket import MINIO_PUBLIC_URL, MINIO_BUCKET, s3 

28from server.utilities import model_parser, sample_payloads 

29from server.validators.query_params_validators import validate_query_params 

30from server.validators.question_request_root_validators import validate_file_size_type 

31from server.validators.question_richtext import enforce_text_length 

32from server.utilities.helpers import question_serializer 

33from server.utilities.sample_payloads import teacher_questionbank_payload 

34from server.utilities.graph_data_checker import ( 

35 attach_graph_fingerprint, 

36 is_graph_question_type, 

37 is_interactive_dots_question_type, 

38) 

39from server.models.question_bank import ( 

40 SOURCE_MAX_LENGTH, 

41 SOURCE_PATTERN, 

42 QuestionModelCreate, 

43 QuestionModelUpdate, 

44) 

45from server.validators.question_class_enum import TypeEnum 

46from datetime import datetime, timezone 

47import re 

48from server.utilities.pagination import resolve_pagination 

49from server.utilities.error_detail import safe_detail 

50 

51_HTML_TAG_RE = re.compile(r"<[^>]+>") 

52 

53 

54def _strip_html_tags(value) -> str: 

55 """Strips HTML tags and collapses whitespace, for comparing a Multi-Group- 

56 Stimulus "drop-down" group's chosen correct answer against its options. 

57 

58 The client stores a chosen answer as plain text (DropdownMenuV2.jsx's 

59 onClick does ``tinyMCEtoString(choice.text)``), while a dropdown's 

60 ``items`` stay the raw rich-text HTML a teacher typed — TinyMCE wraps 

61 even a bare "test 1" in a ``<p>``. A verbatim comparison between the two 

62 would reject every non-trivial answer. 

63 """ 

64 return re.sub(r"\s+", " ", _HTML_TAG_RE.sub(" ", str(value or ""))).strip() 

65 

66 

67def _resolved_pagination(qp) -> dict: 

68 """Parse and bound page/pageSize from the raw query params. 

69 

70 Both were a bare int() with no bounds. Two failure modes came out of that, 

71 live-verified: a non-numeric value raised ValueError, and an out-of-range 

72 one reached the aggregation's $skip/$limit — and BOTH were swallowed by the 

73 broad `except Exception` around the caller, which returned 200 with an empty 

74 result set. A malformed request reported itself as "no questions found", 

75 indistinguishable from a genuinely empty bank. 

76 

77 Parsing here makes a bad value a 400 the caller can act on, and caps 

78 page_size=999999, which returned 9.5 MB. 

79 """ 

80 try: 

81 page = int(qp.get("page", 1)) 

82 except (TypeError, ValueError): 

83 raise HTTPException( 

84 status_code=400, detail="Page number must be a whole number." 

85 ) 

86 try: 

87 page_size = int(qp.get("pageSize", qp.get("page_size", 10))) 

88 except (TypeError, ValueError): 

89 raise HTTPException(status_code=400, detail="Page size must be a whole number.") 

90 

91 resolved_page, resolved_size = resolve_pagination( 

92 page, page_size, default_page_size=10 

93 ) 

94 return {"page": resolved_page, "page_size": resolved_size} 

95 

96 

97class TeacherQuestionService: 

98 """ 

99 Service class for managing question-related operations. 

100 

101 Handles creation, retrieval, updating, and deletion of questions, 

102 as well as question statistics and filtering functionality. 

103 """ 

104 

105 def __init__(self): 

106 pass 

107 

108 # Question images are PUBLIC by design: they are rendered inside a question for 

109 # every student who opens it and are served through the CDN, so they go to 

110 # MINIO_BUCKET (public-read). MINIO_PRIVATE_BUCKET is for PII such as profile 

111 # photos and is only ever read back through short-lived presigned URLs. 

112 QUESTION_IMAGE_PREFIX = "question-images" 

113 

114 async def upload_question_image(self, request: Request, file: UploadFile) -> dict: 

115 """Store an image for embedding in a question and return its public URL. 

116 

117 The editor previously inlined images as base64 ``data:`` URIs, which put the 

118 whole image inside the saved question HTML — re-sent in full to every student 

119 on every attempt, with no CDN caching. This stores the bytes once and hands 

120 back a URL instead. 

121 

122 Args: 

123 request (Request): carries the authenticated teacher on ``request.state``. 

124 file (UploadFile): the image. PNG or JPEG, 10MB max. 

125 

126 Returns: 

127 dict: ``success``, ``message``, ``id``, ``url``, ``path``, 

128 ``original_filename``, ``filename``, ``size`` and ``content_type``. 

129 

130 Raises: 

131 HTTPException: 415 unsupported/undetectable type, 413 too large (both from 

132 ``validate_file_size_type``), or 500 if object storage rejects the upload. 

133 """ 

134 teacher_id = to_user_id(request.state.user_details["uuid"]) 

135 

136 file_obj = file.file 

137 file_obj.seek(0) 

138 # Checks the MAGIC BYTES, not the declared content type, so a .exe renamed 

139 # to .png is rejected here rather than served from the CDN later. 

140 validate_file_size_type(file_obj) 

141 

142 file_obj.seek(0, os.SEEK_END) 

143 size = file_obj.tell() 

144 file_obj.seek(0) 

145 

146 # Trust the sniffed type over the client's claim for the same reason: this 

147 # value becomes the object's Content-Type and therefore how a browser treats 

148 # what it downloads. 

149 detected = filetype.guess(file_obj) 

150 file_obj.seek(0) 

151 content_type = detected.mime if detected else "application/octet-stream" 

152 

153 original_filename = os.path.basename(file.filename or "image") 

154 safe_name = ( 

155 re.sub(r"[^A-Za-z0-9._-]", "_", original_filename).strip("._") or "image" 

156 ) 

157 stored_filename = ( 

158 f"{datetime.now(timezone.utc).strftime('%Y%m%d_%H%M%S')}" 

159 f"_{uuid.uuid4().hex[:8]}_{safe_name}" 

160 ) 

161 # Keyed per teacher: one teacher's uploads can never collide with another's, 

162 # and the owner of an object is readable from its key during cleanup. 

163 key = f"{self.QUESTION_IMAGE_PREFIX}/{teacher_id}/{stored_filename}" 

164 

165 try: 

166 s3.upload_fileobj( 

167 file_obj, 

168 MINIO_BUCKET, 

169 key, 

170 ExtraArgs={"ContentType": content_type}, 

171 ) 

172 except Exception as exc: 

173 logging.error( 

174 f"Question image upload failed for teacher {teacher_id}: {exc}" 

175 ) 

176 raise HTTPException( 

177 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

178 detail="Failed to store the image. Please try again.", 

179 ) 

180 

181 return { 

182 "success": True, 

183 "message": "Image uploaded successfully.", 

184 "id": stored_filename, 

185 "url": f"{MINIO_PUBLIC_URL.rstrip('/')}/{MINIO_BUCKET}/{key}", 

186 "path": key, 

187 "original_filename": original_filename, 

188 "filename": stored_filename, 

189 "size": size, 

190 "content_type": content_type, 

191 } 

192 

193 async def fetch(self, request: Request, teacher_id: str) -> dict: 

194 """ 

195 Fetch paginated teacher questions with comprehensive filtering options. 

196 Supports multiple values for question_types, assignment_types, and category. 

197 Also returns counts grouped by assignmentType, questionType, category, and difficulty. 

198 """ 

199 # Bound before the try: the handler below reads search_params, and the 

200 # very first statement inside the try is the call that assigns it. If that 

201 # call raises (e.g. a malformed query string), the handler would otherwise 

202 # die with UnboundLocalError and mask the original error. 

203 search_params: dict = {} 

204 try: 

205 # Step 1: Extract and normalize query parameters 

206 search_params = self._extract_query_params(request) 

207 

208 # Step 2: Build MongoDB filter criteria 

209 search_filters = { 

210 "createdBy": ObjectId(teacher_id), 

211 "$or": [{"deleted": False}, {"deleted": {"$exists": False}}], 

212 } 

213 

214 # Multi-value, like every other facet (EI-T41 — was single-value, and read 

215 # a parameter name the route does not declare). 

216 if search_params["difficulties"]: 

217 search_filters["difficulty"] = {"$in": search_params["difficulties"]} 

218 

219 # Multi-value filters 

220 if search_params["question_types"]: 

221 search_filters["questionType"] = { 

222 "$in": search_params["question_types"] 

223 } 

224 

225 if search_params["assignment_types"]: 

226 search_filters["assignmentType"] = { 

227 "$in": search_params["assignment_types"] 

228 } 

229 

230 if search_params["categories"]: 

231 search_filters["category"] = {"$in": search_params["categories"]} 

232 

233 if search_params["grade_levels"]: 

234 search_filters["gradeLevel"] = {"$in": search_params["grade_levels"]} 

235 

236 if search_params["subjects"]: 

237 search_filters["questionSubject"] = {"$in": search_params["subjects"]} 

238 

239 self._apply_search_text( 

240 search_filters, search_params.get("search_text", "") 

241 ) 

242 

243 # Pagination parameters 

244 page = search_params["page"] 

245 page_size = search_params["page_size"] 

246 skip = (page - 1) * page_size 

247 

248 # Step 3: Aggregation pipeline to fetch paginated data + grouped counts 

249 pipeline = [ 

250 {"$match": search_filters}, 

251 { 

252 "$facet": { 

253 "questions": [ 

254 {"$sort": {"_id": -1}}, 

255 {"$skip": skip}, 

256 {"$limit": page_size}, 

257 ], 

258 "totalCount": [{"$count": "count"}], 

259 "assignmentTypes": [ 

260 {"$group": {"_id": "$assignmentType", "count": {"$sum": 1}}} 

261 ], 

262 "questionTypes": [ 

263 {"$group": {"_id": "$questionType", "count": {"$sum": 1}}} 

264 ], 

265 "categories": [ 

266 {"$group": {"_id": "$category", "count": {"$sum": 1}}} 

267 ], 

268 "difficulties": [ 

269 {"$group": {"_id": "$difficulty", "count": {"$sum": 1}}} 

270 ], 

271 "gradeLevels": [ 

272 {"$group": {"_id": "$gradeLevel", "count": {"$sum": 1}}} 

273 ], 

274 "subjects": [ 

275 { 

276 "$group": { 

277 "_id": "$questionSubject", 

278 "count": {"$sum": 1}, 

279 } 

280 } 

281 ], 

282 } 

283 }, 

284 ] 

285 

286 results = ( 

287 await db["teacher_questionbank"].aggregate(pipeline).to_list(length=1) 

288 ) 

289 if not results: 

290 # No results found, return empty response 

291 return self._get_empty_response(page, page_size) 

292 

293 result = results[0] 

294 

295 # Deserialize questions using your serializer function 

296 questions = [question_serializer(q) for q in result.get("questions", [])] 

297 

298 # Convert grouped count arrays to dicts for easy consumption 

299 def to_dict(grouped_list): 

300 return { 

301 item["_id"]: item["count"] 

302 for item in grouped_list 

303 if item["_id"] is not None 

304 } 

305 

306 total_count = result.get("totalCount") 

307 total = total_count[0]["count"] if total_count else 0 

308 total_pages = (total + page_size - 1) // page_size 

309 

310 return { 

311 "data": { 

312 "questions": questions, 

313 "assignmentTypes": to_dict(result.get("assignmentTypes", [])), 

314 "questionTypes": to_dict(result.get("questionTypes", [])), 

315 "categories": to_dict(result.get("categories", [])), 

316 "difficulties": to_dict(result.get("difficulties", [])), 

317 "gradeLevels": to_dict(result.get("gradeLevels", [])), 

318 "subjects": to_dict(result.get("subjects", [])), 

319 }, 

320 "pagination": { 

321 "page": page, 

322 "pageSize": page_size, 

323 "totalCount": len(questions), 

324 "totalQuestions": total, 

325 "totalPages": total_pages, 

326 "hasMore": page < total_pages, 

327 }, 

328 } 

329 

330 except HTTPException: 

331 # A 400 from pagination parsing is an ANSWER, not a failure to 

332 # fetch. Without this the broad handler below turned it into 

333 # 200-with-no-questions — a malformed request reporting itself as an 

334 # empty question bank. Its sibling staff_questions_fetch already 

335 # guards this way; this one did not. 

336 raise 

337 except Exception as error: 

338 print(f"Error during data fetching: {error}") 

339 return self._get_empty_response( 

340 search_params.get("page", 1), search_params.get("page_size", 10) 

341 ) 

342 

343 async def staff_questions_fetch(self, request: Request) -> dict: 

344 """ 

345 Fetch paginated staff questions from the admin_staff_mongodb database. 

346 Supports multiple values for question_types, assignment_types, and category. 

347 Also returns counts grouped by assignmentType, questionType, category, and difficulty. 

348 """ 

349 # Bound before the try for the same reason as fetch() above: the handler 

350 # reads search_params, and it is only assigned partway into the try — so a 

351 # 503 from the staff_admin_db check, or a raise inside 

352 # _extract_query_params, would leave it unbound and mask the real error. 

353 search_params: dict = {} 

354 try: 

355 # Check if staff admin database connection is available 

356 if staff_admin_db is None: 

357 raise HTTPException( 

358 status_code=status.HTTP_503_SERVICE_UNAVAILABLE, 

359 detail="Staff Admin database connection not available", 

360 ) 

361 

362 # Step 1: Extract and normalize query parameters 

363 search_params = self._extract_query_params(request) 

364 

365 # Step 2: Build MongoDB filter criteria 

366 search_filters = { 

367 "$or": [{"deleted": False}, {"deleted": {"$exists": False}}], 

368 } 

369 

370 # Multi-value, like every other facet (EI-T41 — was single-value, and read 

371 # a parameter name the route does not declare). 

372 if search_params["difficulties"]: 

373 search_filters["difficulty"] = {"$in": search_params["difficulties"]} 

374 

375 # Multi-value filters 

376 if search_params["question_types"]: 

377 search_filters["questionType"] = { 

378 "$in": search_params["question_types"] 

379 } 

380 

381 if search_params["assignment_types"]: 

382 search_filters["assignmentType"] = { 

383 "$in": search_params["assignment_types"] 

384 } 

385 

386 if search_params["categories"]: 

387 search_filters["category"] = {"$in": search_params["categories"]} 

388 

389 if search_params["grade_levels"]: 

390 search_filters["gradeLevel"] = {"$in": search_params["grade_levels"]} 

391 

392 if search_params["subjects"]: 

393 search_filters["questionSubject"] = {"$in": search_params["subjects"]} 

394 

395 self._apply_search_text( 

396 search_filters, search_params.get("search_text", "") 

397 ) 

398 

399 # Pagination parameters 

400 page = search_params["page"] 

401 page_size = search_params["page_size"] 

402 skip = (page - 1) * page_size 

403 

404 # Step 3: Aggregation pipeline to fetch paginated data + grouped counts 

405 pipeline = [ 

406 {"$match": search_filters}, 

407 { 

408 "$facet": { 

409 "questions": [ 

410 {"$sort": {"_id": -1}}, 

411 {"$skip": skip}, 

412 {"$limit": page_size}, 

413 ], 

414 "totalCount": [{"$count": "count"}], 

415 "assignmentTypes": [ 

416 {"$group": {"_id": "$assignmentType", "count": {"$sum": 1}}} 

417 ], 

418 "questionTypes": [ 

419 {"$group": {"_id": "$questionType", "count": {"$sum": 1}}} 

420 ], 

421 "categories": [ 

422 {"$group": {"_id": "$category", "count": {"$sum": 1}}} 

423 ], 

424 "difficulties": [ 

425 {"$group": {"_id": "$difficulty", "count": {"$sum": 1}}} 

426 ], 

427 "gradeLevels": [ 

428 {"$group": {"_id": "$gradeLevel", "count": {"$sum": 1}}} 

429 ], 

430 "subjects": [ 

431 { 

432 "$group": { 

433 "_id": "$questionSubject", 

434 "count": {"$sum": 1}, 

435 } 

436 } 

437 ], 

438 } 

439 }, 

440 ] 

441 

442 # Fetch from staff admin database's global_questionbank collection 

443 results = ( 

444 await staff_admin_db["global_questionbank"] 

445 .aggregate(pipeline) 

446 .to_list(length=1) 

447 ) 

448 if not results: 

449 # No results found, return empty response 

450 return self._get_empty_response(page, page_size) 

451 

452 result = results[0] 

453 

454 # Deserialize questions using your serializer function 

455 questions = [question_serializer(q) for q in result.get("questions", [])] 

456 

457 # Convert grouped count arrays to dicts for easy consumption 

458 def to_dict(grouped_list): 

459 return { 

460 item["_id"]: item["count"] 

461 for item in grouped_list 

462 if item["_id"] is not None 

463 } 

464 

465 total_count = result.get("totalCount") 

466 total = total_count[0]["count"] if total_count else 0 

467 total_pages = (total + page_size - 1) // page_size 

468 

469 return { 

470 "data": { 

471 "questions": questions, 

472 "assignmentTypes": to_dict(result.get("assignmentTypes", [])), 

473 "questionTypes": to_dict(result.get("questionTypes", [])), 

474 "categories": to_dict(result.get("categories", [])), 

475 "difficulties": to_dict(result.get("difficulties", [])), 

476 "gradeLevels": to_dict(result.get("gradeLevels", [])), 

477 "subjects": to_dict(result.get("subjects", [])), 

478 }, 

479 "pagination": { 

480 "page": page, 

481 "pageSize": page_size, 

482 "totalCount": len(questions), 

483 "totalQuestions": total, 

484 "totalPages": total_pages, 

485 "hasMore": page < total_pages, 

486 }, 

487 } 

488 

489 except HTTPException: 

490 raise 

491 except Exception as error: 

492 print(f"Error during staff questions data fetching: {error}") 

493 return self._get_empty_response( 

494 search_params.get("page", 1), search_params.get("page_size", 10) 

495 ) 

496 

497 @staticmethod 

498 def _apply_search_text(search_filters: dict, search_text: str) -> None: 

499 """Add a case-insensitive free-text match across the searchable fields. 

500 

501 WHY THIS EXISTS 

502 --------------- 

503 The frontend has always sent `searchText`, and this API has never read 

504 it. QuestionBank.jsx then filtered the RESULTS IT ALREADY HAD: 

505 

506 filteredQuestions = (questionData || []).filter(...) 

507 

508 `questionData` is one page — ten questions. So Question Bank search only 

509 ever searched the current page: a teacher with 467 questions in a 

510 category saw "No results" for a term matched by questions two pages 

511 away. Measured on QA 2026-08-17, `searchText` changed nothing at all — 

512 total=5096 with and without it. 

513 

514 Searching the same fields the frontend's client-side filter named, so 

515 behaviour is unchanged where it used to work and merely extends to the 

516 whole result set. `correctAnswer` is matched on its real subfields: 

517 the client did `entry["correctAnswer"].toString()`, which on an object 

518 yields "[object Object]" and never matched anything. 

519 

520 The term is regex-ESCAPED — a teacher typing "2 + 2" or "(x)" must get a 

521 literal search, not a malformed pattern or a catastrophic backtrack. 

522 """ 

523 if not search_text: 

524 return 

525 

526 pattern = {"$regex": re.escape(search_text), "$options": "i"} 

527 clause = { 

528 "$or": [ 

529 {"question": pattern}, 

530 {"questionDetails": pattern}, 

531 {"questionTopic": pattern}, 

532 {"assignmentType": pattern}, 

533 {"questionType": pattern}, 

534 {"teksCode": pattern}, 

535 {"correctAnswer.answers": pattern}, 

536 {"correctAnswer.answerDetails": pattern}, 

537 ] 

538 } 

539 

540 # $and, NOT a second top-level "$or" — search_filters already carries an 

541 # "$or" for the soft-delete check, and assigning another would silently 

542 # REPLACE it, making deleted questions searchable. 

543 search_filters.setdefault("$and", []).append(clause) 

544 

545 def _extract_query_params(self, request: Request) -> dict: 

546 """ 

547 Extracts query parameters including support for multiple values. 

548 """ 

549 qp = request.query_params 

550 

551 assignment_types = ( 

552 getattr(request.state, "assignment_types", None) 

553 or qp.getlist("assignment_types") 

554 or qp.getlist("assignmentType") 

555 or [] 

556 ) 

557 

558 question_types = ( 

559 getattr(request.state, "question_types", None) 

560 or qp.getlist("question_types") 

561 or qp.getlist("questionType") 

562 or [] 

563 ) 

564 

565 categories = ( 

566 getattr(request.state, "categories", None) 

567 or qp.getlist("categories") 

568 or qp.getlist("category") 

569 or [] 

570 ) 

571 

572 # Modified by Allan Ninal — 2026-09-25 (EI-T41) 

573 # WAS: `difficulty = qp.get("difficulty")`. 

574 # Two defects in that one line, both measured live on 0.0.0.377 against the 

575 # endpoint's OWN facet counts: 

576 # 1. `difficulties` — the parameter this route DECLARES and documents, and 

577 # which the handler stores on request.state — was never read here. Sending 

578 # `?difficulties=Easy` returned all 9,212 questions instead of 4,656: a 

579 # documented filter that silently did nothing. 

580 # 2. `qp.get()` returns ONE value. The SPA sends the multi-select as repeated 

581 # `difficulty=` params (QuestionBank.jsx / SearchQuestions.jsx build 

582 # `difficulty: selectedDifficulties`), so ticking Easy AND Average returned 

583 # 2,300 — Average alone, the LAST value — and the other selection was 

584 # dropped with nothing shown to the teacher. 

585 # Every other facet was already a list and multi-selected correctly (control: 

586 # subjects=Math&subjects=Science -> 1,096 = 1,045 + 51), so this reads the same 

587 # way as its five neighbours instead of being the one exception. 

588 difficulties = ( 

589 getattr(request.state, "difficulties", None) 

590 or qp.getlist("difficulties") 

591 or qp.getlist("difficulty") 

592 or [] 

593 ) 

594 

595 grade_levels = ( 

596 getattr(request.state, "grade_levels", None) 

597 or qp.getlist("grade_levels") 

598 or qp.getlist("gradeLevel") 

599 or [] 

600 ) 

601 grade_levels = [int(level) for level in grade_levels] 

602 

603 subjects = ( 

604 getattr(request.state, "subjects", None) 

605 or qp.getlist("subjects") 

606 or qp.getlist("subject") 

607 or [] 

608 ) 

609 

610 # Free-text search. The frontend sends `searchText` (QuestionBank.jsx 

611 # buildQueryString); `search_text` is accepted too for consistency with 

612 # the snake_case aliases above. 

613 search_text = (qp.get("search_text") or qp.get("searchText") or "").strip() 

614 

615 return { 

616 "question_types": question_types, 

617 "assignment_types": assignment_types, 

618 "categories": categories, 

619 "difficulties": difficulties, 

620 "grade_levels": grade_levels, 

621 "subjects": subjects, 

622 "search_text": search_text, 

623 **_resolved_pagination(qp), 

624 } 

625 

626 def _get_empty_response(self, page_number: int, items_per_page: int) -> dict: 

627 """ 

628 Empty response fallback. 

629 """ 

630 return { 

631 "data": { 

632 "questions": [], 

633 "assignmentTypes": {e.value: 0 for e in TypeEnum}, 

634 "questionTypes": {}, 

635 "categories": {}, 

636 "difficulties": {}, 

637 "gradeLevels": {}, 

638 "subjects": {}, 

639 }, 

640 "pagination": { 

641 "page": page_number, 

642 "pageSize": items_per_page, 

643 "totalCount": 0, 

644 "totalQuestions": 0, 

645 "totalPages": 0, 

646 "hasMore": False, 

647 }, 

648 } 

649 

650 @staticmethod 

651 def _mc_choices_or_answers(data: dict): 

652 """Normalised (choices, answers) of a question, for the EI-840 "changed?" test. 

653 

654 Handles every checked type: text choices (Multiple-choice, Checkbox), 

655 id choices (Embedded-MC) and per-blank items (Drop-down-Menu). 

656 """ 

657 

658 def _norm(value): 

659 if isinstance(value, dict): 

660 return {k: _norm(v) for k, v in sorted(value.items(), key=str)} 

661 if isinstance(value, list): 

662 return [_norm(v) for v in value] 

663 return normalize_answer_text(value) 

664 

665 choices = data.get("choices") 

666 answers = (data.get("correctAnswer") or {}).get("answers") 

667 return ( 

668 ( 

669 [ 

670 _norm(c) if isinstance(c, dict) else normalize_answer_text(c) 

671 for c in choices 

672 ] 

673 if isinstance(choices, list) 

674 else None 

675 ), 

676 [_norm(a) for a in answers] if isinstance(answers, list) else None, 

677 ) 

678 

679 @staticmethod 

680 def _check_mc_answer_matches_choice(data: dict) -> None: 

681 """Raise ValueError if a correct answer matches none of its choices. 

682 

683 Multiple-choice / Checkbox: answer text vs choice text (lenient). 

684 Embedded-Multiple-Choice: answer id vs choice id (string equality). 

685 Drop-down-Menu: each blank's answer vs THAT blank's items (lenient); 

686 the blank is the choice whose id equals the answer's id. 

687 """ 

688 question_type = str(data.get("questionType") or "").lower() 

689 answers = (data.get("correctAnswer") or {}).get("answers") 

690 choices = data.get("choices") 

691 if not isinstance(answers, list) or not isinstance(choices, list): 

692 return 

693 # EI-840 (Allan Ninal, 2026-10-04): the answer must match a choice for Checkbox / Embedded-MC / Drop-down too. 

694 if question_type in ("multiple-choice", "checkbox"): 

695 bad = first_unmatched_answer(answers, choices) 

696 if bad is not None: 

697 raise ValueError( 

698 f"Correct answer '{bad}' does not match any of the choices" 

699 ) 

700 elif question_type == "embedded-multiple-choice": 

701 ids = {str(c.get("id")) for c in choices if isinstance(c, dict)} 

702 for answer in answers: 

703 if isinstance(answer, (str, int)) and str(answer) not in ids: 

704 raise ValueError( 

705 f"Correct answer id '{answer}' is not one of the choices" 

706 ) 

707 elif question_type == "drop-down-menu": 

708 blanks: dict = {} 

709 for c in choices: 

710 if isinstance(c, dict) and isinstance(c.get("items"), list): 

711 # A duplicated blank id is ambiguous: leave that blank unchecked. 

712 blanks[str(c.get("id"))] = ( 

713 None if str(c.get("id")) in blanks else c["items"] 

714 ) 

715 for answer in answers: 

716 if not isinstance(answer, dict) or not isinstance( 

717 answer.get("answer"), str 

718 ): 

719 continue 

720 items = blanks.get(str(answer.get("id"))) 

721 if items is None: 

722 continue # unknown or ambiguous blank id: cannot map, skip 

723 if first_unmatched_answer([answer["answer"]], items) is not None: 

724 raise ValueError( 

725 f"Blank {answer.get('id')}: answer '{answer['answer']}' is not one of its items" 

726 ) 

727 

728 async def _validate_question_data( 

729 self, question_data: dict, enforce_mc_match: bool = True 

730 ) -> None: 

731 """ 

732 Validate question data against field rules and template structure. 

733 

734 This method performs comprehensive validation of question data including: 

735 - Question type validation 

736 - Difficulty level validation (must be one of: Easy, Average, Advance; case insensitive) 

737 - Template structure validation 

738 - Required fields validation 

739 - Field type validation 

740 - Field length validation 

741 - Question-type specific validations 

742 

743 Args: 

744 question_data (dict): The question data to validate, containing: 

745 - questionType (str): Type of question (Multiple-choice, Checkbox, etc.) 

746 - difficulty (str): Difficulty level (Easy, Average, Advance; case insensitive) 

747 - question (str): The question text 

748 - correctAnswer (dict): Correct answer information 

749 - questionDetails (str, optional): Additional question details 

750 - assignmentType (str): Type of assignment 

751 - teksCode (str): TEKS code reference 

752 - points (float): Question points (1-100) 

753 - category (str): Question category 

754 - questionTopic (str): Topic of the question 

755 - choices (list, optional): Required for Multiple-choice, Checkbox, Drop-down-Menu 

756 - Other optional fields as defined in field_validations 

757 

758 Raises: 

759 ValueError: If any validation fails, with specific error messages for: 

760 - Invalid question type 

761 - Invalid difficulty level 

762 - Unexpected fields 

763 - Missing required fields 

764 - Invalid field types 

765 - Invalid field lengths 

766 - Invalid choice structures 

767 - Invalid points value 

768 - Other field-specific validations 

769 

770 Examples: 

771 >>> # Valid Multiple-choice question 

772 >>> await _validate_question_data({ 

773 ... "questionType": "Multiple-choice", 

774 ... "difficulty": "Easy", 

775 ... "question": "What is 2+2?", 

776 ... "choices": [{"id": 0, "text": "4"}, {"id": 1, "text": "5"}], 

777 ... "correctAnswer": {"answers": ["4"]}, 

778 ... "questionTopic": "Addition", 

779 ... # ... other required fields ... 

780 ... }) 

781 

782 >>> # Invalid difficulty 

783 >>> await _validate_question_data({ 

784 ... "questionType": "Multiple-choice", 

785 ... "difficulty": "Medium", # Will raise ValueError 

786 ... # ... other fields ... 

787 ... }) 

788 ValueError: Invalid difficulty level. Must be one of: Easy, Average, Advance 

789 

790 Notes: 

791 - All text fields have minimum and maximum length requirements 

792 - Points must be between 1 and 100 with up to 2 decimal places 

793 - Different question types have different validation rules for choices 

794 - Optional fields are only validated if present 

795 """ 

796 # Add valid difficulty levels (lowercase for comparison) 

797 VALID_DIFFICULTY_LEVELS = {"easy", "average", "advance"} 

798 

799 # First validate the question type since other validations depend on it 

800 question_type = question_data.get("questionType") 

801 if not question_type: 

802 raise ValueError("Question Type is required") 

803 

804 # Validate assignmentType against TypeEnum 

805 assignment_type = question_data.get("assignmentType") 

806 if not assignment_type: 

807 raise ValueError("Assignment Type is required") 

808 

809 valid_assignment_types = [e.value for e in TypeEnum] 

810 if assignment_type not in valid_assignment_types: 

811 raise ValueError( 

812 f"Invalid assignment type. Must be one of: {', '.join(valid_assignment_types)}" 

813 ) 

814 

815 # Validate difficulty field (case insensitive) 

816 # Defensive: when the key is present with a None value, .get(key, "") 

817 # returns None (not the default) and .lower() raises AttributeError — 

818 # leaking a 500 to the client. Coerce None to "" so the empty-value 

819 # branch below routes the failure to the controlled 

820 # "Difficulty is required and cannot be empty" 400 at line 549-553. 

821 difficulty = (question_data.get("difficulty") or "").lower() 

822 if difficulty and difficulty not in VALID_DIFFICULTY_LEVELS: 

823 raise ValueError( 

824 "Invalid difficulty level. Must be one of: Advance, Easy, Average" 

825 ) 

826 

827 # Normalize difficulty to proper case if it's valid 

828 if difficulty: 

829 question_data["difficulty"] = difficulty.capitalize() 

830 

831 # Validate template structure early to fail fast if the basic structure is wrong 

832 template_values = [ 

833 teacher_questionbank_payload[template]["value"] 

834 for template in [ 

835 "Multiple-choice", 

836 "Checkbox", 

837 "Free-response", 

838 "Graph", 

839 "Drop-down-Menu", 

840 "Drag-and-Drop", 

841 "Embedded-Multiple-Choice", 

842 "Single-Stimulus", 

843 "Multi-Part-Question", 

844 "Grid-Question", 

845 "Graph-Multiple-Select", 

846 ] 

847 ] 

848 

849 # Get the template for the current question type 

850 current_template = None 

851 for template in template_values: 

852 if template.get("questionType") == question_type: 

853 current_template = template 

854 break 

855 

856 if not current_template: 

857 raise ValueError(f"Invalid question type: {question_type}") 

858 

859 # Validate that SAT, TSI, and ACT tests do not contain releaseDate, category, and teksCode fields 

860 assignment_type = question_data.get("assignmentType") 

861 if assignment_type in ["SAT", "TSI", "ACT"]: 

862 prohibited_fields = ["releaseDate", "category", "teksCode"] 

863 for field in prohibited_fields: 

864 if field in question_data and question_data[field] is not None: 

865 raise ValueError( 

866 f"The field '{field}' is not allowed for {assignment_type} tests. Please remove this field from your request." 

867 ) 

868 

869 # Check for unexpected fields in question_data 

870 allowed_fields = set(current_template.keys()) 

871 unexpected_fields = set(question_data.keys()) - allowed_fields 

872 

873 # Create a list of additional valid fields that might not be in the template 

874 additional_valid_fields = [ 

875 "questionSubject", 

876 "questionGraphs", 

877 "gradeLevel", 

878 "keywords", 

879 "questionImages", 

880 "studentExpectation", 

881 "releaseDate", 

882 "rowHeaderLabel", # Grid-Question only — the first column's header text, defaults to "Statement" 

883 ] 

884 

885 # Remove the additional valid fields from the unexpected_fields set 

886 unexpected_fields = unexpected_fields - set(additional_valid_fields) 

887 

888 if unexpected_fields: 

889 raise ValueError(f"Unexpected fields found: {', '.join(unexpected_fields)}") 

890 

891 # Check if all required keys from template exist in question_data 

892 required_keys = current_template.keys() 

893 

894 # For non-STAAR tests, remove category and teksCode from required template keys 

895 if assignment_type in ["SAT", "TSI", "ACT"]: 

896 required_keys = [ 

897 key for key in required_keys if key not in {"category", "teksCode"} 

898 ] 

899 

900 missing_keys = [key for key in required_keys if key not in question_data] 

901 if missing_keys: 

902 raise ValueError(f"Missing required fields: {', '.join(missing_keys)}") 

903 

904 # Check if the data types match the template 

905 # Special handling for fields that support both string and erudition-math document format 

906 FLEXIBLE_TEXT_FIELDS = {"question", "questionDetails"} 

907 

908 for key, template_value in current_template.items(): 

909 question_value = question_data.get(key) 

910 if question_value is None: 

911 continue 

912 

913 # Allow both string and dict/object for flexible text fields 

914 if key in FLEXIBLE_TEXT_FIELDS: 

915 if not isinstance(question_value, (str, dict)): 

916 raise ValueError( 

917 f"Invalid type for {key}. Expected string or erudition-math document object, " 

918 f"got {type(question_value).__name__}" 

919 ) 

920 # Standard type checking for other fields 

921 elif not isinstance(question_value, type(template_value)): 

922 raise ValueError( 

923 f"Invalid type for {key}. Expected {type(template_value).__name__}, " 

924 f"got {type(question_value).__name__}" 

925 ) 

926 

927 # The "max is ignored" entry for `question` below means this service never 

928 # capped the stem, so a 2,000,000-character question was stored. Apply the 

929 # same plain-vs-rich cap the Pydantic question models use (1,000 plain / 

930 # 500,000 rich editor HTML); empty stems keep their own "required" error. 

931 if isinstance(question_data.get("question"), str): 

932 enforce_text_length(question_data["question"], "question content") 

933 

934 # Validate required fields based on question type 

935 base_required_fields = [ 

936 "question", 

937 "assignmentType", 

938 "questionType", 

939 "difficulty", 

940 "points", 

941 "questionTopic", 

942 ] 

943 

944 # Add category and teksCode field only for STAAR tests 

945 if assignment_type == "STAAR": 

946 base_required_fields.append("category") 

947 base_required_fields.append("teksCode") 

948 

949 # Single-Stimulus and Multi-Part-Question have no top-level 

950 # correctAnswer/choices — each group carries its own (validated below 

951 # via the "groups" branch), so these are the only types that don't 

952 # require "correctAnswer.answers". 

953 if question_type in ["Single-Stimulus", "Multi-Part-Question"]: 

954 required_fields = base_required_fields + ["groups"] 

955 elif question_type == "Grid-Question": 

956 # "choices" (columns) and "rows" (statements) must both be 

957 # structurally valid BEFORE "correctAnswer.answers" is checked, 

958 # since that check cross-references row ids and column text — 

959 # ordered ahead of it here so a malformed rows/choices list gets 

960 # its own specific error instead of a confusing referential one. 

961 required_fields = base_required_fields + [ 

962 "choices", 

963 "rows", 

964 "correctAnswer.answers", 

965 ] 

966 elif question_type in [ 

967 "Multiple-choice", 

968 "Checkbox", 

969 "Drop-down-Menu", 

970 "Drag-and-Drop", 

971 "Embedded-Multiple-Choice", 

972 ]: 

973 required_fields = base_required_fields + [ 

974 "correctAnswer.answers", 

975 "choices", 

976 ] 

977 else: 

978 required_fields = base_required_fields + ["correctAnswer.answers"] 

979 

980 # Custom empty validation rules 

981 for req_field in required_fields: 

982 if "." in req_field: 

983 parent, child = req_field.split(".") 

984 value = question_data.get(parent, {}).get(child) 

985 else: 

986 value = question_data.get(req_field) 

987 

988 # Special validation for correctAnswer.answers based on question type 

989 if req_field == "correctAnswer.answers": 

990 if not value: 

991 raise ValueError("Correct answer is required") 

992 

993 if question_type == "Drag-and-Drop": 

994 # Drag-and-Drop expects a list of objects with id and answer 

995 if not isinstance(value, list) or len(value) == 0: 

996 raise ValueError( 

997 "At least one correct answer must be provided for Drag-and-Drop questions" 

998 ) 

999 for idx, answer in enumerate(value, 1): 

1000 if ( 

1001 not isinstance(answer, dict) 

1002 or "id" not in answer 

1003 or "answer" not in answer 

1004 ): 

1005 raise ValueError( 

1006 f"Answer {idx} must have both 'id' and 'answer' fields" 

1007 ) 

1008 

1009 elif question_type == "Grid-Question": 

1010 # Grid-Question expects one {id, answer} entry per row: id 

1011 # is the row's id, answer is the correct column's TEXT 

1012 # (same {id, answer} shape Drag-and-Drop uses). Unlike 

1013 # Drag-and-Drop, a grid has no "optional" blank — every row 

1014 # must have exactly one correct answer, and every answer 

1015 # must reference a real row and a real column, or the row 

1016 # would be silently ungradable. 

1017 if not isinstance(value, list) or len(value) == 0: 

1018 raise ValueError( 

1019 "At least one correct answer must be provided for Grid-Question questions" 

1020 ) 

1021 for idx, answer in enumerate(value, 1): 

1022 if ( 

1023 not isinstance(answer, dict) 

1024 or "id" not in answer 

1025 or "answer" not in answer 

1026 ): 

1027 raise ValueError( 

1028 f"Answer {idx} must have both 'id' and 'answer' fields" 

1029 ) 

1030 row_ids = { 

1031 row.get("id") 

1032 for row in question_data.get("rows", []) 

1033 if isinstance(row, dict) 

1034 } 

1035 column_texts = { 

1036 choice.get("text") 

1037 for choice in question_data.get("choices", []) 

1038 if isinstance(choice, dict) 

1039 } 

1040 answer_row_ids = [answer.get("id") for answer in value] 

1041 if set(answer_row_ids) != row_ids or len(answer_row_ids) != len( 

1042 row_ids 

1043 ): 

1044 raise ValueError( 

1045 "Every row must have exactly one correct answer, and every answer must reference a row" 

1046 ) 

1047 for answer in value: 

1048 if answer.get("answer") not in column_texts: 

1049 raise ValueError( 

1050 f"Correct answer '{answer.get('answer')}' does not match any column option" 

1051 ) 

1052 

1053 elif question_type in [ 

1054 "Multiple-choice", 

1055 "Checkbox", 

1056 "Embedded-Multiple-Choice", 

1057 ]: 

1058 # Multiple-choice and Checkbox expect a list of strings (answer 

1059 # text); Embedded-Multiple-Choice also expects a list of strings, 

1060 # but each string is the correct choice's `id` rather than its 

1061 # text, since the same phrase can be marked more than once in 

1062 # the passage. 

1063 if not isinstance(value, list) or len(value) == 0: 

1064 raise ValueError("At least one correct answer must be provided") 

1065 

1066 else: # Free-response, Graph 

1067 # Support both plain string and erudition-math document format 

1068 if isinstance(value, list): 

1069 if len(value) == 0: 

1070 raise ValueError("Correct answer is required") 

1071 elif isinstance(value, dict): 

1072 # Erudition-math document format - check for blocks 

1073 if not value.get("blocks") or len(value.get("blocks", [])) == 0: 

1074 raise ValueError("Correct answer is required") 

1075 elif not str(value).strip(): 

1076 raise ValueError("Correct answer is required") 

1077 

1078 # Special validation for choices 

1079 elif req_field == "choices": 

1080 if question_type == "Drop-down-Menu": 

1081 if not value or (isinstance(value, list) and len(value) < 1): 

1082 raise ValueError( 

1083 "For Drop-down-Menu questions, at least one dropdown must be provided" 

1084 ) 

1085 elif question_type == "Embedded-Multiple-Choice": 

1086 if not value or (isinstance(value, list) and len(value) < 1): 

1087 raise ValueError( 

1088 "For Embedded-Multiple-Choice questions, at least one answer choice must be marked in the passage" 

1089 ) 

1090 elif question_type == "Grid-Question": 

1091 if not value or (isinstance(value, list) and len(value) < 2): 

1092 raise ValueError( 

1093 f"For {question_type} questions, at least two choices must be provided" 

1094 ) 

1095 column_texts = [ 

1096 choice.get("text") if isinstance(choice, dict) else None 

1097 for choice in value 

1098 ] 

1099 if len(set(column_texts)) != len(column_texts): 

1100 raise ValueError("Grid-Question columns must have unique text") 

1101 elif not value or (isinstance(value, list) and len(value) < 2): 

1102 raise ValueError( 

1103 f"For {question_type} questions, at least two choices must be provided" 

1104 ) 

1105 

1106 # Special validation for Grid-Question's rows — every row needs a 

1107 # present id (matched against correctAnswer.answers above) and 

1108 # non-empty statement text, and ids must be unique. 

1109 elif req_field == "rows": 

1110 if not isinstance(value, list) or len(value) < 2: 

1111 raise ValueError("A Grid Question needs at least 2 rows") 

1112 row_ids = set() 

1113 for idx, row in enumerate(value, 1): 

1114 if not isinstance(row, dict) or "id" not in row: 

1115 raise ValueError(f"Row {idx}: id is required") 

1116 if row["id"] in row_ids: 

1117 raise ValueError(f"Row {idx}: duplicate row id") 

1118 row_ids.add(row["id"]) 

1119 if not str(row.get("text") or "").strip(): 

1120 raise ValueError(f"Row {idx}: statement text cannot be empty") 

1121 

1122 # Special validation for Single-Stimulus/Multi-Part-Question's 

1123 # groups — each group carries its own type/choices/correctAnswer/ 

1124 # points, independent of every other group's (spec: 1 group = 1 

1125 # independently-scored part). Multi-Part-Question additionally 

1126 # requires every group's own `questionText` (see the check right 

1127 # after group_type below) — its only difference from 

1128 # Single-Stimulus. 

1129 elif req_field == "groups": 

1130 if not isinstance(value, list) or len(value) < 2: 

1131 raise ValueError("A multi-group question needs at least 2 groups") 

1132 

1133 CHOICE_GROUP_TYPES = {"multiple-choice", "checkbox"} 

1134 SUPPORTED_GROUP_TYPES = CHOICE_GROUP_TYPES | { 

1135 "free-response", 

1136 "graph", 

1137 "drop-down", 

1138 "drag-and-drop", 

1139 } 

1140 group_ids = [] 

1141 pages = [] 

1142 for idx, group in enumerate(value, 1): 

1143 if not isinstance(group, dict): 

1144 raise ValueError(f"Group {idx} must be an object") 

1145 

1146 group_id = group.get("group_id") 

1147 if not group_id: 

1148 raise ValueError(f"Group {idx}: group_id is required") 

1149 group_ids.append(group_id) 

1150 pages.append(group.get("page")) 

1151 

1152 group_type = group.get("type") 

1153 if group_type not in SUPPORTED_GROUP_TYPES: 

1154 raise ValueError( 

1155 f"Group {idx}: unsupported group type '{group_type}'. " 

1156 f"Must be one of: {', '.join(sorted(SUPPORTED_GROUP_TYPES))}" 

1157 ) 

1158 

1159 # Multi-Part-Question's one addition over Single-Stimulus: 

1160 # every group also needs its own question text, regardless 

1161 # of the group's type. 

1162 if question_type == "Multi-Part-Question": 

1163 question_text = group.get("questionText") 

1164 if not str(question_text or "").strip(): 

1165 raise ValueError(f"Group {idx}: please add a question") 

1166 

1167 if group_type in CHOICE_GROUP_TYPES: 

1168 group_choices = group.get("choices") 

1169 if ( 

1170 not isinstance(group_choices, list) 

1171 or len(group_choices) < 2 

1172 ): 

1173 raise ValueError( 

1174 f"Group {idx}: at least 2 choices are required" 

1175 ) 

1176 

1177 choice_texts = set() 

1178 for c_idx, choice in enumerate(group_choices, 1): 

1179 text = ( 

1180 choice.get("text", "") 

1181 if isinstance(choice, dict) 

1182 else "" 

1183 ) 

1184 if not str(text).strip(): 

1185 raise ValueError( 

1186 f"Group {idx}, choice {c_idx}: text cannot be empty" 

1187 ) 

1188 choice_texts.add(str(text)) 

1189 if len(choice_texts) != len(group_choices): 

1190 raise ValueError(f"Group {idx}: duplicate choices detected") 

1191 

1192 group_correct = (group.get("correctAnswer") or {}).get( 

1193 "answers" 

1194 ) 

1195 if ( 

1196 not isinstance(group_correct, list) 

1197 or len(group_correct) == 0 

1198 ): 

1199 raise ValueError( 

1200 f"Group {idx}: at least one correct answer must be marked" 

1201 ) 

1202 for answer in group_correct: 

1203 if str(answer) not in choice_texts: 

1204 raise ValueError( 

1205 f"Group {idx}: correct answer '{answer}' does not match any of its choices" 

1206 ) 

1207 elif group_type == "drop-down": 

1208 # A "drop-down" group has its own sentence-with-blanks 

1209 # `content` — separate from the shared main `question`, 

1210 # since the main question has no per-group concept of 

1211 # its own — plus one `choices` entry per blank 

1212 # ({id, items}, same shape the standalone Drop-down-Menu 

1213 # question type uses) and a matching {id, answer} in 

1214 # correctAnswer.answers for every blank (see 

1215 # DropdownMenuV2.jsx / SingleStimulusEditor.jsx on 

1216 # the client). 

1217 content = group.get("content") 

1218 if not str(content or "").strip(): 

1219 raise ValueError(f"Group {idx}: content is required") 

1220 

1221 group_choices = group.get("choices") 

1222 if ( 

1223 not isinstance(group_choices, list) 

1224 or len(group_choices) == 0 

1225 ): 

1226 raise ValueError( 

1227 f"Group {idx}: at least one drop-down blank is required" 

1228 ) 

1229 

1230 blank_items_by_id = {} 

1231 for c_idx, choice in enumerate(group_choices, 1): 

1232 if not isinstance(choice, dict) or "id" not in choice: 

1233 raise ValueError( 

1234 f"Group {idx}, dropdown {c_idx}: must be an object with 'id' and 'items'" 

1235 ) 

1236 items = choice.get("items") 

1237 if not isinstance(items, list) or len(items) < 2: 

1238 raise ValueError( 

1239 f"Group {idx}, dropdown {c_idx}: at least 2 response options are required" 

1240 ) 

1241 for item_idx, item in enumerate(items, 1): 

1242 if not isinstance(item, str) or not item.strip(): 

1243 raise ValueError( 

1244 f"Group {idx}, dropdown {c_idx}, option {item_idx}: cannot be empty" 

1245 ) 

1246 # The client stores a CHOSEN answer as plain text 

1247 # (DropdownMenuV2.jsx's onClick does 

1248 # tinyMCEtoString(choice.text)), while `items` stay 

1249 # the raw rich-text HTML a teacher typed (TinyMCE 

1250 # wraps even a bare "test 1" in a <p>). Comparing 

1251 # the two verbatim below would reject every real 

1252 # answer, so index this blank's options by their 

1253 # stripped plain text instead of the raw HTML. 

1254 blank_items_by_id[choice["id"]] = { 

1255 _strip_html_tags(item) for item in items 

1256 } 

1257 

1258 group_correct = (group.get("correctAnswer") or {}).get( 

1259 "answers" 

1260 ) 

1261 if ( 

1262 not isinstance(group_correct, list) 

1263 or len(group_correct) == 0 

1264 ): 

1265 raise ValueError( 

1266 f"Group {idx}: please select a correct answer for each drop-down" 

1267 ) 

1268 

1269 answered_ids = set() 

1270 for answer in group_correct: 

1271 if not isinstance(answer, dict) or "id" not in answer: 

1272 raise ValueError( 

1273 f"Group {idx}: each drop-down answer must have an 'id' and 'answer'" 

1274 ) 

1275 blank_id = answer.get("id") 

1276 answer_text = answer.get("answer") 

1277 if not str(answer_text or "").strip(): 

1278 raise ValueError( 

1279 f"Group {idx}: please select a correct answer for each drop-down" 

1280 ) 

1281 items = blank_items_by_id.get(blank_id) 

1282 if ( 

1283 items is None 

1284 or _strip_html_tags(answer_text) not in items 

1285 ): 

1286 raise ValueError( 

1287 f"Group {idx}: correct answer for dropdown '{blank_id}' does not match " 

1288 "any of its options" 

1289 ) 

1290 answered_ids.add(blank_id) 

1291 

1292 if set(blank_items_by_id) - answered_ids: 

1293 raise ValueError( 

1294 f"Group {idx}: please select a correct answer for each drop-down" 

1295 ) 

1296 

1297 elif group_type == "drag-and-drop": 

1298 # A "drag-and-drop" group has its own sentence-with-blanks 

1299 # `content` — separate from the shared main `question`, 

1300 # same as "drop-down" above — plus ONE shared pool of 

1301 # draggable choices (flat {id, text}, same shape the 

1302 # standalone Drag-and-Drop question type uses) and a 

1303 # {id, answer} entry per blank in correctAnswer.answers, 

1304 # id === position (see DragDrop.jsx / 

1305 # SingleStimulusEditor.jsx on the client). Unlike 

1306 # "drop-down", the answer stored on a blank is the 

1307 # choice's text VERBATIM — both sides are raw rich-text 

1308 # HTML — so no plain-text normalization is needed here; 

1309 # answer_checking.py's is_answer_correct already grades 

1310 # this correctly through its generic id-map comparison. 

1311 content = group.get("content") 

1312 if not str(content or "").strip(): 

1313 raise ValueError(f"Group {idx}: content is required") 

1314 

1315 group_choices = group.get("choices") 

1316 # The standalone Drag-and-Drop question type requires 

1317 # >= 2 choices (via the generic req_field == "choices" 

1318 # branch above), but its own authoring UI 

1319 # (DragAndDropEditor.jsx) only ever enforces >= 1 — same 

1320 # as this group type's own editor 

1321 # (SingleStimulusEditor.jsx). Mirroring what the UI 

1322 # actually guarantees here, rather than the standalone's 

1323 # stricter and incidental minimum, avoids rejecting a 

1324 # payload the editor itself considers complete — the 

1325 # exact class of bug the "drop-down" group type's own 

1326 # registration fixed after shipping. 

1327 if ( 

1328 not isinstance(group_choices, list) 

1329 or len(group_choices) == 0 

1330 ): 

1331 raise ValueError( 

1332 f"Group {idx}: please add at least one choice" 

1333 ) 

1334 

1335 for c_idx, choice in enumerate(group_choices, 1): 

1336 if ( 

1337 not isinstance(choice, dict) 

1338 or "id" not in choice 

1339 or "text" not in choice 

1340 ): 

1341 raise ValueError( 

1342 f"Group {idx}, choice {c_idx}: must be an object with 'id' and 'text'" 

1343 ) 

1344 if not str(choice.get("text") or "").strip(): 

1345 raise ValueError( 

1346 f"Group {idx}, choice {c_idx}: text cannot be empty" 

1347 ) 

1348 

1349 group_correct = (group.get("correctAnswer") or {}).get( 

1350 "answers" 

1351 ) 

1352 if ( 

1353 not isinstance(group_correct, list) 

1354 or len(group_correct) == 0 

1355 ): 

1356 raise ValueError( 

1357 f"Group {idx}: please add at least one blank in the group's content" 

1358 ) 

1359 

1360 for a_idx, answer in enumerate(group_correct, 1): 

1361 if ( 

1362 not isinstance(answer, dict) 

1363 or "id" not in answer 

1364 or "answer" not in answer 

1365 ): 

1366 raise ValueError( 

1367 f"Group {idx}, blank {a_idx}: must have both 'id' and 'answer' fields" 

1368 ) 

1369 

1370 else: 

1371 # free-response / graph groups carry their answer in 

1372 # correctAnswer.answers directly — same shape (and same 

1373 # leniency) as a standalone Free-response/Graph question's 

1374 # top-level correctAnswer.answers (see the "else" branch 

1375 # above for req_field == "correctAnswer.answers"). 

1376 group_correct = (group.get("correctAnswer") or {}).get( 

1377 "answers" 

1378 ) 

1379 if isinstance(group_correct, list): 

1380 if len(group_correct) == 0: 

1381 raise ValueError( 

1382 f"Group {idx}: a correct answer is required" 

1383 ) 

1384 elif isinstance(group_correct, dict): 

1385 if ( 

1386 not group_correct.get("blocks") 

1387 or len(group_correct.get("blocks", [])) == 0 

1388 ): 

1389 raise ValueError( 

1390 f"Group {idx}: a correct answer is required" 

1391 ) 

1392 elif not str(group_correct or "").strip(): 

1393 raise ValueError( 

1394 f"Group {idx}: a correct answer is required" 

1395 ) 

1396 

1397 group_points = group.get("points", 1) 

1398 try: 

1399 if int(group_points) < 1: 

1400 raise ValueError( 

1401 f"Group {idx}: points must be a positive integer" 

1402 ) 

1403 except (TypeError, ValueError): 

1404 raise ValueError( 

1405 f"Group {idx}: points must be a positive integer" 

1406 ) 

1407 

1408 if len(set(group_ids)) != len(group_ids): 

1409 raise ValueError("Group ids must be unique") 

1410 if len(set(pages)) != len(pages): 

1411 raise ValueError("Group pages must be unique") 

1412 

1413 # General validation for other fields 

1414 elif value is None or str(value).strip() == "": 

1415 raw_field = req_field.split(".")[-1] 

1416 # Convert camelCase or PascalCase to "Title Case" with spaces 

1417 field_name = ( 

1418 re.sub(r"(?<!^)(?=[A-Z])", " ", raw_field).replace("_", " ").title() 

1419 ) 

1420 raise ValueError(f"{field_name} is required and cannot be empty") 

1421 

1422 # Define validation rules 

1423 field_validations = { 

1424 # Required fields with length constraints 

1425 "question": {"min": 5, "max": 2000}, # max is ignored 

1426 "correctAnswer.answers": {"min": 1, "max": 1000}, # max is ignored 

1427 "correctAnswer.answerDetails": { 

1428 "min": 1, 

1429 "max": 2000, 

1430 "required": False, 

1431 }, # max is ignored, optional 

1432 "questionDetails": { 

1433 "min": 5, 

1434 "max": 1000, 

1435 "required": False, 

1436 }, # max is ignored, optional 

1437 "assignmentType": {"min": 2, "max": 20}, # max is ignored 

1438 "questionType": {"min": 2, "max": 20}, # max is ignored 

1439 "difficulty": {"min": 2, "max": 15}, # max is ignored 

1440 "points": {"min": 1, "max": 2}, 

1441 "questionTopic": {"min": 2, "max": 100}, # Now required 

1442 # Optional fields 

1443 "questionImages": {"type": "list", "required": False}, 

1444 "questionGraphs": {"type": "list", "required": False}, 

1445 "correctAnswer.graph": { 

1446 "min": 0, 

1447 "max": 2000, 

1448 "required": False, 

1449 }, # max is ignored 

1450 "studentExpectation": { 

1451 "min": 2, 

1452 "max": 1000, 

1453 "required": False, 

1454 }, # max is ignored 

1455 "questionSubject": { 

1456 "min": 2, 

1457 "max": 100, 

1458 "required": False, 

1459 }, # max is ignored 

1460 "releaseDate": {"type": "datetime", "required": False}, 

1461 "gradeLevel": { 

1462 "type": "integer", 

1463 "min_value": 1, 

1464 "max_value": 12, 

1465 "required": False, 

1466 }, 

1467 "keywords": {"type": "list", "required": False}, 

1468 } 

1469 

1470 # Add question-type specific validations 

1471 if question_type in ["Multiple-choice", "Checkbox"]: 

1472 field_validations["choices"] = { 

1473 "min_items": 2, 

1474 "max_items": 8, # max is ignored 

1475 "text": {"min": 1, "max": 1000}, # max is ignored 

1476 } 

1477 elif question_type == "Drop-down-Menu": 

1478 field_validations["choices"] = { 

1479 "min_items": 1, 

1480 "max_items": 8, # max is ignored 

1481 "items": { 

1482 "min": 2, 

1483 "max": 8, # max is ignored 

1484 "length": {"min": 1, "max": 100}, # max is ignored 

1485 }, 

1486 } 

1487 elif question_type == "Embedded-Multiple-Choice": 

1488 # Flat {id, text} choices, same shape as Multiple-choice/Checkbox, but a 

1489 # passage can reasonably have more marked choices than a typical MC list, 

1490 # and only one needs to exist (validated separately above). 

1491 field_validations["choices"] = { 

1492 "min_items": 1, 

1493 "max_items": 20, # max is ignored 

1494 "text": {"min": 1, "max": 200}, # max is ignored 

1495 } 

1496 

1497 # Add category and teksCode validation - required for STAAR, optional for others 

1498 if assignment_type == "STAAR": 

1499 field_validations["category"] = {"min": 1, "max": 30} # Required for STAAR 

1500 field_validations["teksCode"] = {"min": 2, "max": 15} 

1501 else: 

1502 field_validations["category"] = { 

1503 "min": 1, 

1504 "max": 30, 

1505 "required": False, 

1506 } # Optional for others 

1507 field_validations["teksCode"] = {"min": 2, "max": 15, "required": False} 

1508 

1509 # Validate field lengths and types 

1510 def validate_field_length(data: dict, field: str, rules: dict) -> None: 

1511 # Handle nested fields (e.g., correctAnswer.answers) 

1512 if "." in field: 

1513 parent, child = field.split(".") 

1514 value = data.get(parent, {}).get(child, "") 

1515 else: 

1516 value = data.get(field, "") 

1517 

1518 # Skip validation for non-required fields that are empty 

1519 if not rules.get("required", True) and (value is None or value == ""): 

1520 return 

1521 

1522 # Required field validation 

1523 if rules.get("required", True) and ( 

1524 value is None or str(value).strip() == "" 

1525 ): 

1526 raw_field = field.split(".")[-1] 

1527 readable_field = re.sub(r"(?<!^)(?=[A-Z])", " ", raw_field).replace( 

1528 "_", " " 

1529 ) 

1530 field_name = readable_field.title() 

1531 raise ValueError(f"{field_name} is required and cannot be empty") 

1532 

1533 # Type validations for special fields 

1534 if rules.get("type") == "list" and value is not None: 

1535 if not isinstance(value, list): 

1536 raise ValueError( 

1537 f"{field.replace('_', ' ').title()} must be a list" 

1538 ) 

1539 return 

1540 

1541 if rules.get("type") == "datetime" and value is not None: 

1542 try: 

1543 datetime.fromisoformat(value.replace("Z", "+00:00")) 

1544 except (ValueError, AttributeError): 

1545 raise ValueError( 

1546 f"{field.replace('_', ' ').title()} must be a valid ISO datetime string" 

1547 ) 

1548 return 

1549 

1550 if rules.get("type") == "integer" and value is not None: 

1551 try: 

1552 int_value = int(value) 

1553 if int_value < rules.get( 

1554 "min_value", float("-inf") 

1555 ) or int_value > rules.get("max_value", float("inf")): 

1556 raise ValueError( 

1557 f"{field.replace('_', ' ').title()} must be between {rules.get('min_value')} and {rules.get('max_value')}" 

1558 ) 

1559 except (ValueError, TypeError): 

1560 raise ValueError( 

1561 f"{field.replace('_', ' ').title()} must be an integer" 

1562 ) 

1563 return 

1564 

1565 # Special handling for choices array 

1566 if field == "choices": 

1567 choices = data.get("choices", []) 

1568 if not isinstance(choices, list): 

1569 raise ValueError("Choices must be provided as a list") 

1570 

1571 # Different validation for Drop-down-Menu 

1572 if data.get("questionType") == "Drop-down-Menu": 

1573 # Check number of dropdowns 

1574 if len(choices) < rules["min_items"]: 

1575 raise ValueError( 

1576 f"Please provide at least {rules['min_items']} dropdown" 

1577 ) 

1578 # Commenting out max items validation 

1579 # if len(choices) > rules["max_items"]: 

1580 # raise ValueError( 

1581 # f"Number of dropdowns cannot exceed {rules['max_items']}" 

1582 # ) 

1583 

1584 # Validate each dropdown 

1585 for idx, choice in enumerate(choices, 1): 

1586 if not isinstance(choice, dict): 

1587 raise ValueError( 

1588 f"Dropdown {idx} must be an object with 'id' and 'items' properties" 

1589 ) 

1590 

1591 if "id" not in choice: 

1592 raise ValueError( 

1593 f"Dropdown {idx} must have an 'id' property" 

1594 ) 

1595 

1596 if "items" not in choice: 

1597 raise ValueError( 

1598 f"Dropdown {idx} must have an 'items' property" 

1599 ) 

1600 

1601 items = choice.get("items", []) 

1602 if not isinstance(items, list): 

1603 raise ValueError( 

1604 f"Dropdown {idx} items must be provided as a list" 

1605 ) 

1606 

1607 if len(items) < rules["items"]["min"]: 

1608 raise ValueError( 

1609 f"Dropdown {idx} must have at least {rules['items']['min']} options" 

1610 ) 

1611 

1612 # Validate items 

1613 for item_idx, item in enumerate(items, 1): 

1614 if not isinstance(item, str): 

1615 raise ValueError( 

1616 f"Dropdown {idx} item {item_idx} must be a string" 

1617 ) 

1618 

1619 if len(str(item).strip()) < 1: 

1620 raise ValueError( 

1621 f"Dropdown {idx} item {item_idx} cannot be empty" 

1622 ) 

1623 # Commenting out max length validation 

1624 # if len(str(item).strip()) > 100: 

1625 # raise ValueError(f"Dropdown {idx} item {item_idx} cannot exceed 100 characters") 

1626 

1627 # Standard validation for Multiple-choice and Checkbox 

1628 else: 

1629 if len(choices) < rules["min_items"]: 

1630 raise ValueError( 

1631 f"Please provide at least {rules['min_items']} answer choices" 

1632 ) 

1633 # Commenting out max items validation 

1634 # if len(choices) > rules["max_items"]: 

1635 # raise ValueError( 

1636 # f"Number of choices cannot exceed {rules['max_items']}" 

1637 # ) 

1638 

1639 for idx, choice in enumerate(choices, 1): 

1640 # A list of bare strings/numbers has no .get(); refuse it 

1641 # as a client error instead of crashing with a 500. 

1642 if not isinstance(choice, dict): 

1643 raise ValueError( 

1644 f"Choice {idx} must be an object with 'id' and 'text' properties, " 

1645 f"got {type(choice).__name__}" 

1646 ) 

1647 text = choice.get("text", "") 

1648 

1649 # Support dual format: both string and erudition-math document format 

1650 if isinstance(text, dict): 

1651 # Erudition-Math document format validation 

1652 if not text.get("blocks") or not isinstance( 

1653 text.get("blocks"), list 

1654 ): 

1655 raise ValueError( 

1656 f"Choice {idx} text document format must have a 'blocks' array" 

1657 ) 

1658 # Skip length validation for document format 

1659 elif isinstance(text, str): 

1660 # Plain text format validation 

1661 if len(text.strip()) < rules["text"]["min"]: 

1662 raise ValueError( 

1663 f"Choice {idx} text is too short. Minimum length is {rules['text']['min']} character" 

1664 ) 

1665 # Commenting out max length validation 

1666 # if len(text.strip()) > rules["text"]["max"]: 

1667 # raise ValueError( 

1668 # f"Choice {idx} text is too long. Maximum length is {rules['text']['max']} characters" 

1669 # ) 

1670 else: 

1671 raise ValueError( 

1672 f"Choice {idx} text must be either a string or erudition-math document object, " 

1673 f"got {type(text).__name__}" 

1674 ) 

1675 return 

1676 

1677 # Special handling for Drop-down-Menu choices 

1678 if ( 

1679 field == "choices_dropdown" 

1680 and data.get("questionType") == "Drop-down-Menu" 

1681 ): 

1682 choices = data.get("choices", []) 

1683 if not isinstance(choices, list): 

1684 raise ValueError( 

1685 "Choices must be provided as a list of dropdown options" 

1686 ) 

1687 

1688 # Check number of dropdowns 

1689 if len(choices) < rules["min_items"]: 

1690 raise ValueError( 

1691 f"Please provide at least {rules['min_items']} dropdown" 

1692 ) 

1693 # Commenting out max items validation 

1694 # if len(choices) > rules["max_items"]: 

1695 # raise ValueError( 

1696 # f"Number of dropdowns cannot exceed {rules['max_items']}" 

1697 # ) 

1698 

1699 # Check each dropdown's structure and options 

1700 for idx, choice in enumerate(choices, 1): 

1701 # Validate choice structure 

1702 if not isinstance(choice, dict): 

1703 raise ValueError( 

1704 f"Dropdown {idx} must be an object with 'id' and 'items' properties" 

1705 ) 

1706 

1707 if "id" not in choice: 

1708 raise ValueError(f"Dropdown {idx} must have an 'id' property") 

1709 

1710 if "items" not in choice: 

1711 raise ValueError( 

1712 f"Dropdown {idx} must have an 'items' property" 

1713 ) 

1714 

1715 items = choice.get("items", []) 

1716 if not isinstance(items, list): 

1717 raise ValueError( 

1718 f"Dropdown {idx} items must be provided as a list" 

1719 ) 

1720 

1721 # Check number of items 

1722 if len(items) < rules["items"]["min"]: 

1723 raise ValueError( 

1724 f"Dropdown {idx} must have at least {rules['items']['min']} options" 

1725 ) 

1726 # Commenting out max items validation 

1727 # if len(items) > rules["items"]["max"]: 

1728 # raise ValueError( 

1729 # f"Dropdown {idx} cannot exceed {rules['items']['max']} options" 

1730 # ) 

1731 

1732 # Check each item's text length 

1733 for option_idx, option in enumerate(items, 1): 

1734 if not isinstance(option, str): 

1735 raise ValueError( 

1736 f"Dropdown {idx} option {option_idx} must be a string" 

1737 ) 

1738 

1739 if len(option.strip()) < rules["items"]["length"]["min"]: 

1740 raise ValueError( 

1741 f"Dropdown {idx} option {option_idx} is too short. Minimum length is {rules['items']['length']['min']} character" 

1742 ) 

1743 # Commenting out max length validation 

1744 # if len(option.strip()) > rules["items"]["length"]["max"]: 

1745 # raise ValueError( 

1746 # f"Dropdown {idx} option {option_idx} is too long. Maximum length is {rules['items']['length']['max']} characters" 

1747 # ) 

1748 return 

1749 

1750 # Handle numeric fields 

1751 if field == "points": 

1752 # EI-2781: cap at 100 (was 99). The AIO contract for EI-TC-1074 requires 

1753 # points=100 to be valid (normal-operation boundary) and EI-TC-1075 

1754 # requires very large values (e.g., 1,000,000) to be rejected. 

1755 try: 

1756 # Convert to float first to handle both integer and decimal 

1757 points = float(value) 

1758 # Check if it's within range (1 to 100) 

1759 if not (1 <= points <= 100): 

1760 raise ValueError("Points must be between 1 and 100") 

1761 # Check if it has more than 2 decimal places 

1762 if ( 

1763 len(str(points).split(".")[-1]) > 2 

1764 if "." in str(points) 

1765 else False 

1766 ): 

1767 raise ValueError("Points can have up to 2 decimal places") 

1768 except (ValueError, TypeError): 

1769 raise ValueError("Points must be a number between 1 and 100") 

1770 return 

1771 

1772 # Skip length validation for erudition-math document format (dict/object) 

1773 if isinstance(value, dict): 

1774 # For document objects, just verify they have valid structure 

1775 if field in [ 

1776 "question", 

1777 "questionDetails", 

1778 "correctAnswer.answers", 

1779 "correctAnswer.answerDetails", 

1780 ]: 

1781 if not value.get("blocks") or not isinstance( 

1782 value.get("blocks"), list 

1783 ): 

1784 raise ValueError( 

1785 f"{field.replace('_', ' ').title()} document format must have a 'blocks' array" 

1786 ) 

1787 return 

1788 

1789 if not isinstance(value, str): 

1790 value = str(value) 

1791 

1792 length = len(value.strip()) 

1793 if length < rules["min"]: 

1794 raise ValueError( 

1795 f"{field.replace('_', ' ').title()} must be at least {rules['min']} characters long" 

1796 ) 

1797 # Commenting out max length validation 

1798 # if length > rules["max"]: 

1799 # raise ValueError( 

1800 # f"{field.replace('_', ' ').title()} cannot exceed {rules['max']} characters" 

1801 # ) 

1802 

1803 # Check all fields. correctAnswer.answers is skipped for 

1804 # Single-Stimulus/Multi-Part-Question — neither has a top-level 

1805 # correct answer by design (each group's correctAnswer.answers was 

1806 # already validated above). 

1807 for field, rules in field_validations.items(): 

1808 if field == "correctAnswer.answers" and question_type in [ 

1809 "Single-Stimulus", 

1810 "Multi-Part-Question", 

1811 ]: 

1812 continue 

1813 validate_field_length(question_data, field, rules) 

1814 

1815 if enforce_mc_match: 

1816 self._check_mc_answer_matches_choice(question_data) 

1817 

1818 def _sanitize_question_rich_text(self, data: dict) -> dict: 

1819 """Neutralise stored XSS in every rich-text field on a question (EI-3321). 

1820 

1821 The STAFF bank has sanitised every write path since EI-2481. The teacher bank 

1822 had no sanitiser at all: a question could be stored with <script> and on*= 

1823 handlers intact, and rendered later straight into another user's page. Proven 

1824 against the model on 2026-09-06 — "<script>alert('XSS');</script>" survived 

1825 byte-for-byte. 

1826 

1827 Applied on create AND update so neither path can be the way round it. Mutates 

1828 and returns `data`. 

1829 """ 

1830 if not isinstance(data, dict): 

1831 return data 

1832 

1833 if isinstance(data.get("question"), str) and data["question"]: 

1834 data["question"] = sanitize_rich_text(data["question"]) 

1835 

1836 for key in ("questionDetails", "solutions"): 

1837 if isinstance(data.get(key), str) and data[key]: 

1838 data[key] = sanitize_rich_text(data[key]) 

1839 

1840 for choice in data.get("choices") or []: 

1841 if not isinstance(choice, dict): 

1842 continue 

1843 if isinstance(choice.get("text"), str) and choice["text"]: 

1844 choice["text"] = sanitize_rich_text(choice["text"]) 

1845 # Drop-down-Menu keeps its options nested one level down. 

1846 for item in choice.get("items") or []: 

1847 if not isinstance(item, dict): 

1848 continue 

1849 for key in ("text", "answer", "answerDetails"): 

1850 if isinstance(item.get(key), str) and item[key]: 

1851 item[key] = sanitize_rich_text(item[key]) 

1852 

1853 # Multi-Part-Question and Single-Stimulus carry their real content one level 

1854 # down, in `groups`. Left out until EI-3321 was extended for question import, 

1855 # every part's questionText and choices reached the bank unsanitised — the 

1856 # exact hole this function exists to close, for the two types where the 

1857 # top-level `question` is only the shared stimulus. 

1858 for group in data.get("groups") or []: 

1859 if not isinstance(group, dict): 

1860 continue 

1861 for key in ("questionText", "content"): 

1862 if isinstance(group.get(key), str) and group[key]: 

1863 group[key] = sanitize_rich_text(group[key]) 

1864 for choice in group.get("choices") or []: 

1865 if not isinstance(choice, dict): 

1866 continue 

1867 if isinstance(choice.get("text"), str) and choice["text"]: 

1868 choice["text"] = sanitize_rich_text(choice["text"]) 

1869 # Indexed, not `list.index(item)`: two identical dropdown options are 

1870 # legitimate, and index() would rewrite the first one twice and leave 

1871 # the duplicate untouched. 

1872 items = choice.get("items") 

1873 for position, item in enumerate(items or []): 

1874 if isinstance(item, str) and item: 

1875 items[position] = sanitize_rich_text(item) 

1876 elif isinstance(item, dict): 

1877 for key in ("text", "answer", "answerDetails"): 

1878 if isinstance(item.get(key), str) and item[key]: 

1879 item[key] = sanitize_rich_text(item[key]) 

1880 group_correct = group.get("correctAnswer") 

1881 if isinstance(group_correct, dict): 

1882 group_answers = group_correct.get("answers") 

1883 if isinstance(group_answers, list): 

1884 for position, item in enumerate(group_answers): 

1885 if isinstance(item, dict) and isinstance( 

1886 item.get("answer"), str 

1887 ): 

1888 item["answer"] = sanitize_rich_text(item["answer"]) 

1889 elif isinstance(item, str): 

1890 group_answers[position] = sanitize_rich_text(item) 

1891 elif isinstance(group_answers, str) and group_answers: 

1892 group_correct["answers"] = sanitize_rich_text(group_answers) 

1893 if ( 

1894 isinstance(group_correct.get("answerDetails"), str) 

1895 and group_correct["answerDetails"] 

1896 ): 

1897 group_correct["answerDetails"] = sanitize_rich_text( 

1898 group_correct["answerDetails"] 

1899 ) 

1900 

1901 correct = data.get("correctAnswer") 

1902 if isinstance(correct, dict): 

1903 # Graph and Graph-Multiple-Select answers are serialised canvases, not 

1904 # rich text — sanitising them would corrupt the Graph2D JSON. 

1905 is_graph = is_graph_question_type(data.get("questionType")) or is_interactive_dots_question_type( 

1906 data.get("questionType") 

1907 ) 

1908 answers = correct.get("answers") 

1909 if not is_graph: 

1910 if isinstance(answers, list): 

1911 for item in answers: 

1912 if isinstance(item, dict) and isinstance( 

1913 item.get("answer"), str 

1914 ): 

1915 item["answer"] = sanitize_rich_text(item["answer"]) 

1916 elif isinstance(item, str): 

1917 answers[answers.index(item)] = sanitize_rich_text(item) 

1918 elif isinstance(answers, str) and answers: 

1919 correct["answers"] = sanitize_rich_text(answers) 

1920 if ( 

1921 isinstance(correct.get("answerDetails"), str) 

1922 and correct["answerDetails"] 

1923 ): 

1924 correct["answerDetails"] = sanitize_rich_text(correct["answerDetails"]) 

1925 

1926 return attach_graph_fingerprint(data) 

1927 

1928 async def create( 

1929 self, 

1930 request: Request, 

1931 question_data: Annotated[ 

1932 dict, 

1933 Body( 

1934 openapi_examples={ 

1935 k: Example(value=v["value"]) 

1936 for k, v in teacher_questionbank_payload.items() 

1937 } 

1938 ), 

1939 ], 

1940 *, 

1941 source: str | None = None, 

1942 ): 

1943 """ 

1944 Create a new question as a teacher. 

1945 

1946 Args: 

1947 request (Request): The incoming request object containing teacher context 

1948 question_data (dict): Question details matching one of the template values 

1949 from teacher_questionbank_payload templates 

1950 source (str | None): How the question entered the bank, recorded for audit. 

1951 Keyword-only and NOT part of question_data on purpose: the field check 

1952 in `_validate_question_data` rejects any key the template does not 

1953 declare, so a `source` key inside the payload made every question-import 

1954 commit fail with "Unexpected fields found: source". Callers that accept 

1955 client-supplied bodies (the create route) leave it unset, so a client 

1956 still cannot claim a provenance for itself. 

1957 

1958 Returns: 

1959 dict: Newly created question details 

1960 

1961 Raises: 

1962 HTTPException: 

1963 - 400 if validation fails (invalid assignmentType, questionType, etc.) 

1964 - 500 if creation fails 

1965 """ 

1966 try: 

1967 await self._validate_question_data(question_data) 

1968 question_data = self._sanitize_question_rich_text(question_data) 

1969 

1970 question_dict = { 

1971 **question_data, 

1972 "createdDate": datetime.now(timezone.utc), 

1973 "createdBy": to_user_id(request.state.user_details["uuid"]), 

1974 } 

1975 

1976 if source is not None: 

1977 # Same rule as QuestionModelCreate.source, imported rather than 

1978 # retyped: it is read by audits and written to log lines, so an 

1979 # arbitrary string could forge a plausible-looking record. 

1980 if ( 

1981 not re.fullmatch(SOURCE_PATTERN, source) 

1982 or len(source) > SOURCE_MAX_LENGTH 

1983 ): 

1984 raise ValueError( 

1985 f"Invalid source. Must match {SOURCE_PATTERN} " 

1986 f"and be at most {SOURCE_MAX_LENGTH} characters" 

1987 ) 

1988 question_dict["source"] = source 

1989 

1990 result = await db["teacher_questionbank"].insert_one(question_dict) 

1991 return { 

1992 "new_question": question_serializer( 

1993 {**question_dict, "_id": result.inserted_id} 

1994 ) 

1995 } 

1996 except ValueError as e: 

1997 # EI-2781: payload validation errors return HTTP 400 (was 422). Aligns 

1998 # with the AIO contract for `/v1/teacher/question/*` and the existing 

1999 # 400 convention used by the global `RequestValidationError` handler 

2000 # for `/v1/teacher/account/*` (EI-TC-995) and `/v1/teacher/class/*` 

2001 # (EI-TC-982). 

2002 raise HTTPException(status.HTTP_400_BAD_REQUEST, detail=str(e)) 

2003 except HTTPException: 

2004 raise 

2005 except Exception as e: 

2006 raise HTTPException( 

2007 status.HTTP_500_INTERNAL_SERVER_ERROR, detail=safe_detail(e) 

2008 ) 

2009 

2010 async def detail_fetch(self, question_id: str, request: Request) -> Dict[str, Any]: 

2011 """ 

2012 Fetch details of a specific teacher question by ID. 

2013 

2014 This endpoint retrieves a single question's complete details, ensuring the 

2015 requesting teacher has appropriate access rights. 

2016 

2017 Parameters 

2018 ---------- 

2019 question_id : str 

2020 Unique identifier of the question to retrieve 

2021 request : Request 

2022 FastAPI request object containing authenticated teacher details 

2023 in request.state.user_details["uuid"] 

2024 

2025 Returns 

2026 ------- 

2027 Dict[str, Any] 

2028 Serialized question details including: 

2029 - id: str 

2030 - question: str 

2031 - questionType: str 

2032 - assignmentType: str 

2033 - difficulty: str 

2034 - category: str 

2035 - points: int 

2036 - choices: List[Dict] (for multiple choice questions) 

2037 - correctAnswer: Dict 

2038 - createdBy: str 

2039 - createdDate: datetime 

2040 

2041 Raises 

2042 ------ 

2043 HTTPException 

2044 400: 

2045 Invalid question ID format provided 

2046 404: 

2047 Question not found or teacher lacks access permission 

2048 

2049 Examples 

2050 -------- 

2051 >>> # Valid request 

2052 >>> question = await detail_fetch("507f1f77bcf86cd799439011", request) 

2053 >>> print(question["questionType"]) 

2054 'multiple-choice' 

2055 

2056 >>> # Invalid ID format 

2057 >>> await detail_fetch("invalid-id", request) 

2058 HTTPException: 400 Bad Request - Invalid question ID format 

2059 

2060 Notes 

2061 ----- 

2062 - Checks both explicit deletion flag and missing deletion flag 

2063 - Verifies teacher ownership of question 

2064 - Returns 404 for both missing questions and unauthorized access 

2065 """ 

2066 # Validate MongoDB ObjectId format 

2067 if not ObjectId.is_valid(question_id): 

2068 raise HTTPException( 

2069 status_code=status.HTTP_400_BAD_REQUEST, 

2070 detail="Invalid question ID format", 

2071 ) 

2072 

2073 # Query for active question owned by requesting teacher 

2074 teacher_question = await db["teacher_questionbank"].find_one( 

2075 { 

2076 "_id": ObjectId(question_id), 

2077 "createdBy": to_user_id(request.state.user_details["uuid"]), 

2078 # Include both non-deleted and never-marked-as-deleted questions 

2079 "$or": [{"deleted": False}, {"deleted": {"$exists": False}}], 

2080 } 

2081 ) 

2082 

2083 # Handle not found or unauthorized access 

2084 if not teacher_question: 

2085 raise HTTPException( 

2086 status_code=status.HTTP_404_NOT_FOUND, 

2087 detail="Question not found or unauthorized access", 

2088 ) 

2089 

2090 # Serialize and return question data 

2091 return question_serializer(teacher_question) 

2092 

2093 async def update( 

2094 self, 

2095 question_id: str, 

2096 question_data: Annotated[ 

2097 dict, 

2098 Body( 

2099 openapi_examples={ 

2100 k: Example(value=v["value"]) 

2101 for k, v in teacher_questionbank_payload.items() 

2102 } 

2103 ), 

2104 ], 

2105 request: Request, 

2106 ) -> dict: 

2107 """ 

2108 Update a question in the database. 

2109 

2110 Parameters: 

2111 ---------- 

2112 question_id : str 

2113 The ID of the question to update 

2114 question_data : dict 

2115 The updated question data 

2116 request : Request 

2117 FastAPI request object containing user context 

2118 

2119 Returns: 

2120 -------- 

2121 dict 

2122 A dictionary containing the updated question under 'updated_question' key 

2123 

2124 Raises: 

2125 ------- 

2126 HTTPException 

2127 400: Invalid question ID format 

2128 404: Question not found or user not authorized 

2129 400: Validation fails (invalid assignmentType, questionType, etc.) 

2130 500: Database operation error 

2131 """ 

2132 if not ObjectId.is_valid(question_id): 

2133 raise HTTPException(status_code=400, detail="Invalid question ID format") 

2134 

2135 try: 

2136 # Provenance is set once, when the question is created, and records how it 

2137 # entered the bank. An edit must not be able to rewrite it -- otherwise a 

2138 # question could later claim to have arrived some other way, and the field 

2139 # an audit reads would be worth nothing. 

2140 # 

2141 # Stripped HERE rather than by leaving it off QuestionModelUpdate, because 

2142 # this endpoint takes a raw dict: `**question_data` below spreads whatever 

2143 # the client sent straight into $set, so the model is not in the path. 

2144 # 

2145 # Stripped BEFORE validation, not after: the field check rejects any key the 

2146 # template does not declare, so validating first turned "this field is 

2147 # ignored on update" into a 400 for every edit that echoed back a question 

2148 # the import created. 

2149 question_data = {k: v for k, v in question_data.items() if k != "source"} 

2150 

2151 # EI-840: create is strict, but an UPDATE is checked only when it 

2152 # changes the choices or the answers (grandfathering: a stored 

2153 # question that already mismatches stays editable as long as the edit 

2154 # resends its choices and answers unchanged). 

2155 await self._validate_question_data(question_data, enforce_mc_match=False) 

2156 

2157 sent_answers = (question_data.get("correctAnswer") or {}).get("answers") 

2158 stored = None 

2159 if ( 

2160 "questionType" in question_data 

2161 and "choices" not in question_data 

2162 and sent_answers is None 

2163 ): 

2164 pass # touches neither choices nor answers 

2165 else: 

2166 stored = await db["teacher_questionbank"].find_one( 

2167 { 

2168 "_id": ObjectId(question_id), 

2169 "createdBy": to_user_id(request.state.user_details["uuid"]), 

2170 "deleted": {"$ne": True}, 

2171 }, 

2172 {"questionType": 1, "choices": 1, "correctAnswer": 1}, 

2173 ) 

2174 if stored: 

2175 effective = {**stored, **question_data} 

2176 if isinstance(question_data.get("correctAnswer"), dict): 

2177 effective["correctAnswer"] = { 

2178 **(stored.get("correctAnswer") or {}), 

2179 **question_data["correctAnswer"], 

2180 } 

2181 if self._mc_choices_or_answers(effective) != self._mc_choices_or_answers( 

2182 stored 

2183 ): 

2184 self._check_mc_answer_matches_choice(effective) 

2185 

2186 question_data = self._sanitize_question_rich_text(question_data) 

2187 

2188 result = await db["teacher_questionbank"].find_one_and_update( 

2189 { 

2190 "_id": ObjectId(question_id), 

2191 "createdBy": to_user_id(request.state.user_details["uuid"]), 

2192 "deleted": {"$ne": True}, 

2193 }, 

2194 { 

2195 "$set": { 

2196 **question_data, 

2197 "updatedDate": datetime.now(timezone.utc), 

2198 "updatedBy": to_user_id(request.state.user_details["uuid"]), 

2199 } 

2200 }, 

2201 return_document=ReturnDocument.AFTER, 

2202 ) 

2203 

2204 if not result: 

2205 raise HTTPException( 

2206 status_code=404, detail="Question not found or unauthorized access" 

2207 ) 

2208 

2209 return {"updated_question": question_serializer(result)} 

2210 except ValueError as e: 

2211 # EI-2781: payload validation errors return HTTP 400 (was 422) for 

2212 # consistency with the create path and the global `/v1/teacher/*/*` 

2213 # 400 convention. 

2214 raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e)) 

2215 except HTTPException: 

2216 raise 

2217 except Exception as e: 

2218 print(f"Error updating question: {str(e)}") 

2219 raise HTTPException( 

2220 status_code=500, detail="Error while updating the question" 

2221 ) 

2222 

2223 async def delete(self, question_id: str, request: Request) -> dict: 

2224 """ 

2225 Soft delete a teacher question. 

2226 

2227 Parameters 

2228 ---------- 

2229 question_id : str 

2230 The unique identifier of the question to delete 

2231 request : Request 

2232 The FastAPI request object containing authenticated user details 

2233 

2234 Returns 

2235 ------- 

2236 dict 

2237 A dictionary containing a success message 

2238 

2239 Raises 

2240 ------ 

2241 HTTPException 

2242 - 400: If the question_id format is invalid 

2243 - 403: If the user is not authorized to delete the question 

2244 - 404: If the question is not found or already deleted 

2245 - 500: If there's a database operation error 

2246 """ 

2247 try: 

2248 # Validate question ID format 

2249 if not ObjectId.is_valid(question_id): 

2250 raise HTTPException( 

2251 status_code=400, detail="Invalid question ID format" 

2252 ) 

2253 

2254 # Perform soft delete operation 

2255 result = await db["teacher_questionbank"].find_one_and_update( 

2256 { 

2257 "_id": ObjectId(question_id), 

2258 "createdBy": to_user_id(request.state.user_details["uuid"]), 

2259 "deleted": {"$ne": True}, 

2260 }, 

2261 { 

2262 "$set": { 

2263 "deleted": True, 

2264 "deletedDate": datetime.now(timezone.utc), 

2265 "deletedBy": to_user_id(request.state.user_details["uuid"]), 

2266 } 

2267 }, 

2268 return_document=ReturnDocument.AFTER, 

2269 ) 

2270 

2271 if not result: 

2272 raise HTTPException( 

2273 status_code=404, detail="Question not found or unauthorized access" 

2274 ) 

2275 

2276 return {"message": "Successfully deleted question"} 

2277 

2278 except HTTPException: 

2279 raise 

2280 except Exception: 

2281 raise HTTPException( 

2282 status_code=500, detail="Error while deleting the question" 

2283 ) 

2284 

2285 async def get_filter_options(self, request: Request) -> dict: 

2286 """ 

2287 Get available filter options with counts. 

2288 

2289 Retrieves counts of questions for each filter option category. 

2290 

2291 Args: 

2292 request (Request): FastAPI request object with authentication context 

2293 

2294 Returns: 

2295 dict: Filter options with counts: 

2296 - assignmentTypes: Count by assignment type 

2297 - questionTypes: Count by question type 

2298 - categories: Count by category 

2299 - difficulties: Count by difficulty 

2300 

2301 Raises: 

2302 HTTPException: When database operations fail 

2303 """ 

2304 try: 

2305 teacher_id = to_user_id(request.state.user_details["uuid"]) 

2306 

2307 # Base filter to get only this teacher's non-deleted questions 

2308 base_filter = { 

2309 "createdBy": ObjectId(teacher_id), 

2310 "$or": [{"deleted": False}, {"deleted": {"$exists": False}}], 

2311 } 

2312 

2313 # Get all assignment types with counts 

2314 assignment_types_pipeline = [ 

2315 {"$match": base_filter}, 

2316 {"$group": {"_id": "$assignmentType", "count": {"$sum": 1}}}, 

2317 {"$sort": {"_id": 1}}, 

2318 ] 

2319 assignment_types_result = ( 

2320 await db["teacher_questionbank"] 

2321 .aggregate(assignment_types_pipeline) 

2322 .to_list(100) 

2323 ) 

2324 assignment_types = { 

2325 item["_id"]: item["count"] 

2326 for item in assignment_types_result 

2327 if item["_id"] 

2328 } 

2329 

2330 # Get all question types with counts 

2331 question_types_pipeline = [ 

2332 {"$match": base_filter}, 

2333 {"$group": {"_id": "$questionType", "count": {"$sum": 1}}}, 

2334 {"$sort": {"_id": 1}}, 

2335 ] 

2336 question_types_result = ( 

2337 await db["teacher_questionbank"] 

2338 .aggregate(question_types_pipeline) 

2339 .to_list(100) 

2340 ) 

2341 question_types = { 

2342 item["_id"]: item["count"] 

2343 for item in question_types_result 

2344 if item["_id"] 

2345 } 

2346 

2347 # Get all categories with counts 

2348 categories_pipeline = [ 

2349 {"$match": base_filter}, 

2350 {"$group": {"_id": "$category", "count": {"$sum": 1}}}, 

2351 {"$sort": {"_id": 1}}, 

2352 ] 

2353 categories_result = ( 

2354 await db["teacher_questionbank"] 

2355 .aggregate(categories_pipeline) 

2356 .to_list(100) 

2357 ) 

2358 categories = { 

2359 item["_id"]: item["count"] for item in categories_result if item["_id"] 

2360 } 

2361 

2362 # Get all difficulties with counts 

2363 difficulties_pipeline = [ 

2364 {"$match": base_filter}, 

2365 {"$group": {"_id": "$difficulty", "count": {"$sum": 1}}}, 

2366 {"$sort": {"_id": 1}}, 

2367 ] 

2368 difficulties_result = ( 

2369 await db["teacher_questionbank"] 

2370 .aggregate(difficulties_pipeline) 

2371 .to_list(100) 

2372 ) 

2373 difficulties = { 

2374 item["_id"]: item["count"] 

2375 for item in difficulties_result 

2376 if item["_id"] 

2377 } 

2378 

2379 return { 

2380 "assignmentTypes": assignment_types, 

2381 "questionTypes": question_types, 

2382 "categories": categories, 

2383 "difficulties": difficulties, 

2384 } 

2385 

2386 except Exception as error: 

2387 print(f"Error getting filter options: {error}") 

2388 raise HTTPException( 

2389 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

2390 detail="Failed to retrieve filter options", 

2391 )