Coverage for server / services / teacher / question_dedup.py: 100%

26 statements  

« prev     ^ index     » next       coverage.py v7.13.4, created at 2026-10-04 09:33 +0000

1"""Duplicate detection for teacher question imports. 

2 

3Ported from the staff implementation (`app/services/staff/question_dedup.py` in 

4mathmatterstx-services), which had itself carried the behaviour over from the legacy CSV 

5bulk upload. The teacher side never got it: a teacher who imported the same file twice 

6silently doubled those questions, with no warning at review and no report afterwards. 

7The staff code's own comment describes exactly that failure as the thing it exists to 

8prevent — the lesson was applied to one portal and not the other. 

9 

10The rule, identical to staff: 

11 

12 fingerprint = (question, questionType) 

13 scope = createdBy == this teacher, and not soft-deleted 

14 

15Scoped to the owner on purpose. Two teachers may legitimately hold the same question; 

16only re-importing *your own* is a duplicate. 

17 

18Deliberately a plain module, not a method: the import route needs it at BOTH review time 

19(so duplicates are visible before anything is written) and commit time (so they are 

20skipped), and neither is a natural home for the other. 

21 

22Developer: Allan Ninal 

23""" 

24 

25from __future__ import annotations 

26 

27import logging 

28 

29from server.connection.database import db 

30 

31logger = logging.getLogger(__name__) 

32 

33COLLECTION = "teacher_questionbank" 

34 

35# A question is a duplicate only against live rows. Live documents here OMIT the field 

36# entirely -- verified on the bank: 16,096 of 34,907 rows carry `deleted: true` and none 

37# of the live ones carry the key at all. So "missing" MUST count as not-deleted; 

38# inverting it would make every live question re-importable. 

39_NOT_DELETED = {"$or": [{"deleted": False}, {"deleted": {"$exists": False}}]} 

40 

41 

42def fingerprint(payload: dict) -> tuple[str, str]: 

43 """The identity used for duplicate comparison. 

44 

45 Accepts BOTH key shapes, because this helper is called at two points that speak 

46 different dialects: 

47 

48 * review -- the ingest service's own model dump, which is snake_case 

49 (`question_type`) 

50 * commit -- the mapped API payload, which is camelCase (`questionType`) 

51 

52 On the staff side, reading only `questionType` made every review-time fingerprint 

53 `(question, "")`, which matched nothing: the review screen always reported zero 

54 duplicates while the commit-time check worked. Silently wrong in the direction that 

55 looks fine -- "no duplicates" is a plausible answer. Do not simplify this. 

56 

57 `.get` with a default throughout: a payload missing the type must not raise, it 

58 should simply fail to match anything. 

59 """ 

60 question = payload.get("question", "") or "" 

61 question_type = payload.get("questionType") or payload.get("question_type") or "" 

62 return (question, question_type) 

63 

64 

65async def existing_fingerprints(teacher_id, payloads: list[dict]) -> set[tuple[str, str]]: 

66 """Fingerprints this teacher already owns, among the ones offered. 

67 

68 One query for the whole batch, narrowed by `question $in`, rather than a lookup per 

69 row: an import is up to 500 questions and per-row queries turn that into 500 

70 round-trips. 

71 

72 Served by the `dedup_createdBy_question` index created at startup 

73 (server/connection/database.py). Without it this is a full collection scan. 

74 """ 

75 unique_texts = sorted({text for text, _ in map(fingerprint, payloads) if text}) 

76 if not unique_texts: 

77 return set() 

78 

79 cursor = db[COLLECTION].find( 

80 { 

81 "createdBy": teacher_id, 

82 "question": {"$in": unique_texts}, 

83 **_NOT_DELETED, 

84 }, 

85 {"question": 1, "questionType": 1, "_id": 0}, 

86 ) 

87 return {fingerprint(doc) async for doc in cursor} 

88 

89 

90async def mark_duplicates( 

91 teacher_id, payloads: list[dict] 

92) -> tuple[list[int], set[tuple[str, str]]]: 

93 """Indices of payloads that already exist, plus the fingerprints that matched. 

94 

95 Also catches duplicates WITHIN the uploaded file, not just against the bank. A file 

96 listing the same question twice would otherwise import it twice on the first run and 

97 be reported as a duplicate only on the second. 

98 """ 

99 existing = await existing_fingerprints(teacher_id, payloads) 

100 seen: set[tuple[str, str]] = set() 

101 duplicate_indices: list[int] = [] 

102 

103 for index, payload in enumerate(payloads): 

104 fp = fingerprint(payload) 

105 if fp in existing or fp in seen: 

106 duplicate_indices.append(index) 

107 else: 

108 seen.add(fp) 

109 

110 return duplicate_indices, existing