Compare commits

..
2 changed files with 11 additions and 65 deletions
+11 -53
View File
@@ -13,7 +13,6 @@ join keys (spec §2).
from __future__ import annotations from __future__ import annotations
import json import json
import math
import os import os
import tempfile import tempfile
import time import time
@@ -343,13 +342,6 @@ def _pdf_has_text_layer(pdf_bytes: bytes) -> bool:
pass pass
# Canvas page width the frontend renders each PDF page at (app src/utils/exam-canvas/model.ts
# PAGE_WIDTH). All auto-map canvas coords are emitted in this 780-wide, proportional-height space.
CANVAS_PAGE_WIDTH = 780.0
# Response/answer-region detector (api/services/docling/regions.py) renders at 144 DPI = 2 px / PDF point.
REGIONS_PX_PER_PT = 2.0
def _pdf_page_geometry(pdf_bytes: bytes) -> List[Dict[str, float]]: def _pdf_page_geometry(pdf_bytes: bytes) -> List[Dict[str, float]]:
with tempfile.NamedTemporaryFile(prefix="cc-auto-map-geom-", suffix=".pdf", delete=False) as fh: with tempfile.NamedTemporaryFile(prefix="cc-auto-map-geom-", suffix=".pdf", delete=False) as fh:
fh.write(pdf_bytes) fh.write(pdf_bytes)
@@ -363,23 +355,14 @@ def _pdf_page_geometry(pdf_bytes: bytes) -> List[Dict[str, float]]:
for page in doc: for page in doc:
media = page.mediabox media = page.mediabox
crop = page.cropbox crop = page.cropbox
page_pt_w = float(crop.width or page.rect.width or 1.0) rendered_w = float(crop.width or page.rect.width or 595.0)
page_pt_h = float(crop.height or page.rect.height or 1.0) rendered_h = float(crop.height or page.rect.height or 842.0)
# Emit canvas coords in the FRONTEND render space: the app draws each page at
# CANVAS_PAGE_WIDTH (app model.ts PAGE_WIDTH=780) with proportional height and stacks
# pages by those heights. Previously rendered_w/h were left in PDF points (~595x842),
# so every shape landed shrunk (~0.76x) and shifted up-left on the 780-wide canvas.
rendered_w = CANVAS_PAGE_WIDTH
# Mirror the app's canvas.height = Math.ceil(viewport.height) EXACTLY (pdfLoader.ts),
# so page_top accumulates identically. Using the raw float drifts ~1px/page, compounding
# to a visible upward shift on later pages of long papers (~36px over 40 pages).
rendered_h = float(math.ceil(CANVAS_PAGE_WIDTH * page_pt_h / page_pt_w))
pages.append({ pages.append({
"media_x0": float(media.x0), "media_x0": float(media.x0),
"crop_x0": float(crop.x0), "crop_x0": float(crop.x0),
"crop_y0": float(crop.y0), "crop_y0": float(crop.y0),
"page_pt_w": page_pt_w, "page_pt_w": float(crop.width or page.rect.width or 1),
"page_pt_h": page_pt_h, "page_pt_h": float(crop.height or page.rect.height or 1),
"rendered_w": rendered_w, "rendered_w": rendered_w,
"rendered_h": rendered_h, "rendered_h": rendered_h,
"page_top": page_top, "page_top": page_top,
@@ -401,12 +384,11 @@ def _pdf_page_geometry(pdf_bytes: bytes) -> List[Dict[str, float]]:
def _page_geom(pages: List[Dict[str, float]], page_number: int) -> Dict[str, float]: def _page_geom(pages: List[Dict[str, float]], page_number: int) -> Dict[str, float]:
if 1 <= page_number <= len(pages): if 1 <= page_number <= len(pages):
return pages[page_number - 1] return pages[page_number - 1]
_fallback_h = float(math.ceil(CANVAS_PAGE_WIDTH * 842.0 / 595.0))
return { return {
"media_x0": 0.0, "crop_x0": 0.0, "crop_y0": 0.0, "media_x0": 0.0, "crop_x0": 0.0, "crop_y0": 0.0,
"page_pt_w": 595.0, "page_pt_h": 842.0, "page_pt_w": 595.0, "page_pt_h": 842.0,
"rendered_w": CANVAS_PAGE_WIDTH, "rendered_h": _fallback_h, "rendered_w": 595.0, "rendered_h": 842.0,
"page_top": (page_number - 1) * _fallback_h, "page_top": (page_number - 1) * 842.0,
} }
@@ -415,16 +397,12 @@ def _box_to_canvas(box: Optional[Dict[str, Any]], page_number: int, pages: List[
return None return None
g = _page_geom(pages, page_number) g = _page_geom(pages, page_number)
if box.get("coord_origin") == "TOPLEFT" and {"x", "y", "w", "h"}.issubset(box): if box.get("coord_origin") == "TOPLEFT" and {"x", "y", "w", "h"}.issubset(box):
# Scale the box into the 780-wide canvas space. px boxes (opencv/gemma regions) are in scale = 0.5 if box.get("unit") == "px" else 1.0
# rendered-image px at REGIONS_PX_PER_PT px/point; TOPLEFT point boxes are 1 px/point.
px_per_pt = REGIONS_PX_PER_PT if box.get("unit") == "px" else 1.0
sx = g["rendered_w"] / (g["page_pt_w"] * px_per_pt)
sy = g["rendered_h"] / (g["page_pt_h"] * px_per_pt)
return { return {
"x": round(float(box["x"]) * sx, 2), "x": round(float(box["x"]) * scale, 2),
"y": round(g["page_top"] + float(box["y"]) * sy, 2), "y": round(g["page_top"] + float(box["y"]) * scale, 2),
"w": round(float(box["w"]) * sx, 2), "w": round(float(box["w"]) * scale, 2),
"h": round(float(box["h"]) * sy, 2), "h": round(float(box["h"]) * scale, 2),
} }
if not {"l", "t", "r", "b"}.issubset(box): if not {"l", "t", "r", "b"}.issubset(box):
return None return None
@@ -562,23 +540,6 @@ def _map_first_pass_to_rows(template_id: str, first_pass: Dict[str, Any], pdf_by
response_form = _response_form_from_region_type(region.get("region_type")) response_form = _response_form_from_region_type(region.get("region_type"))
if response_form: if response_form:
response_areas.append({"id": _ai_id(template_id, "region", page_index, idx), "template_id": template_id, "question_id": first_part_by_page.get(page_index, default_qid), "page": page_index + 1, "bounds": bounds, "kind": "response", "response_form": response_form, "source": "ai", "confirmed": False, "confidence": _safe_confidence(region.get("confidence")), "derivation": region.get("detection_method") or "opencv-response-region"}) response_areas.append({"id": _ai_id(template_id, "region", page_index, idx), "template_id": template_id, "question_id": first_part_by_page.get(page_index, default_qid), "page": page_index + 1, "bounds": bounds, "kind": "response", "response_form": response_form, "source": "ai", "confirmed": False, "confidence": _safe_confidence(region.get("confidence")), "derivation": region.get("detection_method") or "opencv-response-region"})
# Integrity guard: every response_area/boundary question_id must reference an inserted question
# (FK exam_response_areas/exam_boundaries -> exam_questions). On papers where band detection yields
# few/no questions but opencv/gemma still emit regions, those regions point at the synthetic
# default_qid which was never inserted. Ensure that fallback container question exists and reattach
# any orphan child rows to it, so persistence can't violate the FK.
qid_set = {q["id"] for q in questions}
orphans = [r for r in (response_areas + boundaries) if r.get("question_id") not in qid_set]
if orphans:
if default_qid not in qid_set:
questions.insert(0, {"id": default_qid, "template_id": template_id, "label": "Unassigned",
"order": 0, "max_marks": 0, "is_container": True, "source": "ai",
"confirmed": False, "confidence": 0.5,
"derivation": "auto-map-fallback-container"})
qid_set.add(default_qid)
for r in orphans:
r["question_id"] = default_qid
return {"questions": questions, "response_areas": response_areas, "boundaries": boundaries, "layout": layout} return {"questions": questions, "response_areas": response_areas, "boundaries": boundaries, "layout": layout}
@@ -634,9 +595,6 @@ def _run_auto_map_job(job_id: str, ctx: ExamContext, template_id: str, pdf_bytes
_set_auto_map_status(job_id, {"status": "running", "template_id": template_id}) _set_auto_map_status(job_id, {"status": "running", "template_id": template_id})
try: try:
rows = _run_auto_map_merge(ctx, template_id, pdf_bytes, source_label) rows = _run_auto_map_merge(ctx, template_id, pdf_bytes, source_label)
# Project to Neo4j like the born-digital fast path does — otherwise image-only papers (R3's
# primary target, routed here because they need OCR) never reach the graph after auto-map.
project_template_safe(template_id)
_set_auto_map_status(job_id, {"status": "completed", "template_id": template_id, "counts": {k: len(v) for k, v in rows.items()}}) _set_auto_map_status(job_id, {"status": "completed", "template_id": template_id, "counts": {k: len(v) for k, v in rows.items()}})
except Exception as exc: except Exception as exc:
logger.exception(f"auto-map job failed for template {template_id}: {exc}") logger.exception(f"auto-map job failed for template {template_id}: {exc}")
-12
View File
@@ -718,16 +718,4 @@ def test_auto_map_ocr_returns_job_id_and_status_completes(monkeypatch):
body = status.json() body = status.json()
assert body["status"] == "completed" assert body["status"] == "completed"
assert body["counts"]["questions"] >= 2 assert body["counts"]["questions"] >= 2
def test_auto_map_ocr_path_projects_to_neo4j(monkeypatch, _stub_projection):
# Image-only papers route through the async OCR job; regression: that path must project to Neo4j
# like the born-digital fast path, or the graph is never built for the primary target.
store = _template_with_source()
client, store = make_client(store=store)
_patch_auto_map(monkeypatch, store, fast=False)
resp = client.post("/api/exam/templates/t1/auto-map")
assert resp.status_code == 202
# the BackgroundTask runs after the response under TestClient
assert "t1" in _stub_projection
assert body["template"]["layout"] assert body["template"]["layout"]