diff --git a/backend/app/tasks/quiz_tasks.py b/backend/app/tasks/quiz_tasks.py index 3d60393..afb6c80 100644 --- a/backend/app/tasks/quiz_tasks.py +++ b/backend/app/tasks/quiz_tasks.py @@ -385,6 +385,165 @@ QUESTIONS DIFFICULTY_BATCH = 25 +#: Judging a figure against its question needs a model that can see. The job +#: model is passed in rather than read from the admin's `tool` setting, because +#: this is a one-off audit and not a standing capability — it should not +#: quietly become the thing that decides what "the tool model" means. +FIGURE_AUDIT_MODEL = "openrouter-gemini-2.5-flash" + +FIGURE_AUDIT_PROMPT = """This figure is attached to the exam question below. Decide whether it belongs +to it. + +It belongs if a candidate would need it, or could reasonably use it, to answer +*this* question — a radiograph of the finding, a growth chart the question asks +you to read, a table of the schedule being asked about. It does not belong if it +is about some other topic entirely, which happens when a figure was lifted from +the wrong page of a source document. + +Be strict about "no" and generous about "unsure": detaching a figure a question +needs is worse than leaving a stray one attached for somebody to notice. + +Answer ONLY with JSON: +{"belongs": "yes" | "no" | "unsure", "shows": "one sentence describing the figure"} + +QUESTION +""" + + +@celery_app.task(name="audit_question_figures", bind=True) +def audit_question_figures(self, job_id: str = "", limit: int | None = None, + detach: bool = True, model_id: str | None = None) -> dict: + """Look at every figure attached to a question and say whether it belongs. + + Extraction took figures off the page a question was printed on, which is + usually right and sometimes catastrophically wrong: an HPV vaccination + question was carrying a table of craniofacial reconstruction by age. Nobody + can find those by reading stems, because the mistake is only visible in the + picture. + + Every figure is also *described* while it is being looked at, and the + description is kept — most of them said "Figure from question #1206", which + is a filename with extra steps and made the image bank unsearchable. + + Detaching is reversible in the sense that matters: the asset stays in the + bank with a note saying what happened, and only the link goes. + """ + import json as _json + + from app.models.media import MediaAsset + from app.models.question import Question + from app.models.question_media import QuestionMedia + from app.services import storage_service, thumbnails, vision_service + from app.services.ai_service import chat + + r = _redis() + if job_id: + r.set(f"extraction:status:{job_id}", "running", ex=EXPIRE_SECONDS) + db = SessionLocal() + model = model_id or FIGURE_AUDIT_MODEL + seen = kept = detached = unsure = unreadable = described = 0 + mismatches: list[dict] = [] + try: + rows = db.query(Question).filter( + Question.image_path.isnot(None), Question.image_path != "").order_by(Question.id) + questions = rows.limit(limit).all() if limit else rows.all() + total = len(questions) + if job_id: + _push_step(r, job_id, "start", f"{total} figures to look at") + + for position, question in enumerate(questions, start=1): + seen += 1 + data = None + try: + data = storage_service.load(question.image_path) + except Exception: + data = None + if not data: + unreadable += 1 + continue + # 640px is what the reader is shown and plenty to judge a figure by; + # sending the original would be megabytes a question for no gain. + small = thumbnails.render(data, 640) or data + image = vision_service.prepare(small, "image/webp", caption="a figure attached to a question") + if image is None: + unreadable += 1 + continue + + stem = (question.question_text or "")[:900] + try: + raw = (chat(model=model, max_tokens=300, temperature=0, messages=[{ + "role": "user", "content": [ + vision_service.image_part(image), + {"type": "text", "text": FIGURE_AUDIT_PROMPT + stem}, + ]}]) or "").strip() + if raw.startswith("```"): + raw = raw.split("\n", 1)[1] if "\n" in raw else raw[3:] + raw = raw[:-3] if raw.endswith("```") else raw + verdict = _json.loads(raw.strip()) + except Exception as exc: + unreadable += 1 + logger.warning("Figure audit failed for question %s: %s", question.id, exc) + continue + + belongs = str(verdict.get("belongs", "unsure")).strip().lower() + shows = str(verdict.get("shows", "")).strip()[:600] + + # The description is worth keeping whatever the verdict: "Figure + # from question #1206" is a filename with extra steps. + asset = db.query(MediaAsset).filter(MediaAsset.path == question.image_path).first() + if asset and shows: + placeholder = not asset.alt_text or asset.alt_text.lower().startswith("figure from question") + if placeholder: + asset.alt_text = shows + described += 1 + if not asset.caption: + asset.caption = shows + + if belongs == "no": + mismatches.append({"question_id": question.id, "path": question.image_path, + "shows": shows, "stem": stem[:160], "verdict": "no"}) + if detach: + question.image_path = None + db.query(QuestionMedia).filter( + QuestionMedia.question_id == question.id).delete(synchronize_session=False) + if asset: + note = f"Detached from question #{question.id}: the figure shows {shows}" + asset.caption = (asset.caption or shows) + asset.alt_text = asset.alt_text or note + detached += 1 + else: + kept += 1 + elif belongs == "unsure": + unsure += 1 + # Named, not just counted: "2 unsure" is a number nobody can + # act on. These stay attached and go on the list for a person. + mismatches.append({"question_id": question.id, "path": question.image_path, + "shows": shows, "stem": stem[:160], "verdict": "unsure"}) + else: + kept += 1 + + db.commit() + if job_id and position % 10 == 0: + _push_step(r, job_id, "batch", + f"{position} of {total} · {detached} detached · {unsure} unsure") + + result = {"looked_at": seen, "kept": kept, "detached": detached, "unsure": unsure, + "unreadable": unreadable, "described": described, + "mismatches": mismatches[:60]} + if job_id: + r.set(f"extraction:status:{job_id}", "completed", ex=EXPIRE_SECONDS) + _push_step(r, job_id, "done", + f"{detached} figures detached, {unsure} to check by hand") + return result + except Exception as exc: + logger.warning("Figure audit failed: %s", exc) + if job_id: + r.set(f"extraction:status:{job_id}", "failed", ex=EXPIRE_SECONDS) + raise + finally: + db.close() + + @celery_app.task(name="classify_question_difficulty", bind=True) def classify_question_difficulty(self, job_id: str = "", limit: int | None = None, relabel: bool = False) -> dict: