Coverage for server / services / teacher / question_dedup.py: 100%
26 statements
« prev ^ index » next coverage.py v7.13.4, created at 2026-10-04 09:33 +0000
« prev ^ index » next coverage.py v7.13.4, created at 2026-10-04 09:33 +0000
1"""Duplicate detection for teacher question imports.
3Ported from the staff implementation (`app/services/staff/question_dedup.py` in
4mathmatterstx-services), which had itself carried the behaviour over from the legacy CSV
5bulk upload. The teacher side never got it: a teacher who imported the same file twice
6silently doubled those questions, with no warning at review and no report afterwards.
7The staff code's own comment describes exactly that failure as the thing it exists to
8prevent — the lesson was applied to one portal and not the other.
10The rule, identical to staff:
12 fingerprint = (question, questionType)
13 scope = createdBy == this teacher, and not soft-deleted
15Scoped to the owner on purpose. Two teachers may legitimately hold the same question;
16only re-importing *your own* is a duplicate.
18Deliberately a plain module, not a method: the import route needs it at BOTH review time
19(so duplicates are visible before anything is written) and commit time (so they are
20skipped), and neither is a natural home for the other.
22Developer: Allan Ninal
23"""
25from __future__ import annotations
27import logging
29from server.connection.database import db
31logger = logging.getLogger(__name__)
33COLLECTION = "teacher_questionbank"
35# A question is a duplicate only against live rows. Live documents here OMIT the field
36# entirely -- verified on the bank: 16,096 of 34,907 rows carry `deleted: true` and none
37# of the live ones carry the key at all. So "missing" MUST count as not-deleted;
38# inverting it would make every live question re-importable.
39_NOT_DELETED = {"$or": [{"deleted": False}, {"deleted": {"$exists": False}}]}
42def fingerprint(payload: dict) -> tuple[str, str]:
43 """The identity used for duplicate comparison.
45 Accepts BOTH key shapes, because this helper is called at two points that speak
46 different dialects:
48 * review -- the ingest service's own model dump, which is snake_case
49 (`question_type`)
50 * commit -- the mapped API payload, which is camelCase (`questionType`)
52 On the staff side, reading only `questionType` made every review-time fingerprint
53 `(question, "")`, which matched nothing: the review screen always reported zero
54 duplicates while the commit-time check worked. Silently wrong in the direction that
55 looks fine -- "no duplicates" is a plausible answer. Do not simplify this.
57 `.get` with a default throughout: a payload missing the type must not raise, it
58 should simply fail to match anything.
59 """
60 question = payload.get("question", "") or ""
61 question_type = payload.get("questionType") or payload.get("question_type") or ""
62 return (question, question_type)
65async def existing_fingerprints(teacher_id, payloads: list[dict]) -> set[tuple[str, str]]:
66 """Fingerprints this teacher already owns, among the ones offered.
68 One query for the whole batch, narrowed by `question $in`, rather than a lookup per
69 row: an import is up to 500 questions and per-row queries turn that into 500
70 round-trips.
72 Served by the `dedup_createdBy_question` index created at startup
73 (server/connection/database.py). Without it this is a full collection scan.
74 """
75 unique_texts = sorted({text for text, _ in map(fingerprint, payloads) if text})
76 if not unique_texts:
77 return set()
79 cursor = db[COLLECTION].find(
80 {
81 "createdBy": teacher_id,
82 "question": {"$in": unique_texts},
83 **_NOT_DELETED,
84 },
85 {"question": 1, "questionType": 1, "_id": 0},
86 )
87 return {fingerprint(doc) async for doc in cursor}
90async def mark_duplicates(
91 teacher_id, payloads: list[dict]
92) -> tuple[list[int], set[tuple[str, str]]]:
93 """Indices of payloads that already exist, plus the fingerprints that matched.
95 Also catches duplicates WITHIN the uploaded file, not just against the bank. A file
96 listing the same question twice would otherwise import it twice on the first run and
97 be reported as a duplicate only on the second.
98 """
99 existing = await existing_fingerprints(teacher_id, payloads)
100 seen: set[tuple[str, str]] = set()
101 duplicate_indices: list[int] = []
103 for index, payload in enumerate(payloads):
104 fp = fingerprint(payload)
105 if fp in existing or fp in seen:
106 duplicate_indices.append(index)
107 else:
108 seen.add(fp)
110 return duplicate_indices, existing