Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
023c1d9b95 | ||
|
|
be347418f6 | ||
|
|
31c51cb7aa |
+11
-53
@@ -13,7 +13,6 @@ join keys (spec §2).
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import tempfile
|
||||
import time
|
||||
@@ -343,13 +342,6 @@ def _pdf_has_text_layer(pdf_bytes: bytes) -> bool:
|
||||
pass
|
||||
|
||||
|
||||
# Canvas page width the frontend renders each PDF page at (app src/utils/exam-canvas/model.ts
|
||||
# PAGE_WIDTH). All auto-map canvas coords are emitted in this 780-wide, proportional-height space.
|
||||
CANVAS_PAGE_WIDTH = 780.0
|
||||
# Response/answer-region detector (api/services/docling/regions.py) renders at 144 DPI = 2 px / PDF point.
|
||||
REGIONS_PX_PER_PT = 2.0
|
||||
|
||||
|
||||
def _pdf_page_geometry(pdf_bytes: bytes) -> List[Dict[str, float]]:
|
||||
with tempfile.NamedTemporaryFile(prefix="cc-auto-map-geom-", suffix=".pdf", delete=False) as fh:
|
||||
fh.write(pdf_bytes)
|
||||
@@ -363,23 +355,14 @@ def _pdf_page_geometry(pdf_bytes: bytes) -> List[Dict[str, float]]:
|
||||
for page in doc:
|
||||
media = page.mediabox
|
||||
crop = page.cropbox
|
||||
page_pt_w = float(crop.width or page.rect.width or 1.0)
|
||||
page_pt_h = float(crop.height or page.rect.height or 1.0)
|
||||
# Emit canvas coords in the FRONTEND render space: the app draws each page at
|
||||
# CANVAS_PAGE_WIDTH (app model.ts PAGE_WIDTH=780) with proportional height and stacks
|
||||
# pages by those heights. Previously rendered_w/h were left in PDF points (~595x842),
|
||||
# so every shape landed shrunk (~0.76x) and shifted up-left on the 780-wide canvas.
|
||||
rendered_w = CANVAS_PAGE_WIDTH
|
||||
# Mirror the app's canvas.height = Math.ceil(viewport.height) EXACTLY (pdfLoader.ts),
|
||||
# so page_top accumulates identically. Using the raw float drifts ~1px/page, compounding
|
||||
# to a visible upward shift on later pages of long papers (~36px over 40 pages).
|
||||
rendered_h = float(math.ceil(CANVAS_PAGE_WIDTH * page_pt_h / page_pt_w))
|
||||
rendered_w = float(crop.width or page.rect.width or 595.0)
|
||||
rendered_h = float(crop.height or page.rect.height or 842.0)
|
||||
pages.append({
|
||||
"media_x0": float(media.x0),
|
||||
"crop_x0": float(crop.x0),
|
||||
"crop_y0": float(crop.y0),
|
||||
"page_pt_w": page_pt_w,
|
||||
"page_pt_h": page_pt_h,
|
||||
"page_pt_w": float(crop.width or page.rect.width or 1),
|
||||
"page_pt_h": float(crop.height or page.rect.height or 1),
|
||||
"rendered_w": rendered_w,
|
||||
"rendered_h": rendered_h,
|
||||
"page_top": page_top,
|
||||
@@ -401,12 +384,11 @@ def _pdf_page_geometry(pdf_bytes: bytes) -> List[Dict[str, float]]:
|
||||
def _page_geom(pages: List[Dict[str, float]], page_number: int) -> Dict[str, float]:
|
||||
if 1 <= page_number <= len(pages):
|
||||
return pages[page_number - 1]
|
||||
_fallback_h = float(math.ceil(CANVAS_PAGE_WIDTH * 842.0 / 595.0))
|
||||
return {
|
||||
"media_x0": 0.0, "crop_x0": 0.0, "crop_y0": 0.0,
|
||||
"page_pt_w": 595.0, "page_pt_h": 842.0,
|
||||
"rendered_w": CANVAS_PAGE_WIDTH, "rendered_h": _fallback_h,
|
||||
"page_top": (page_number - 1) * _fallback_h,
|
||||
"rendered_w": 595.0, "rendered_h": 842.0,
|
||||
"page_top": (page_number - 1) * 842.0,
|
||||
}
|
||||
|
||||
|
||||
@@ -415,16 +397,12 @@ def _box_to_canvas(box: Optional[Dict[str, Any]], page_number: int, pages: List[
|
||||
return None
|
||||
g = _page_geom(pages, page_number)
|
||||
if box.get("coord_origin") == "TOPLEFT" and {"x", "y", "w", "h"}.issubset(box):
|
||||
# Scale the box into the 780-wide canvas space. px boxes (opencv/gemma regions) are in
|
||||
# rendered-image px at REGIONS_PX_PER_PT px/point; TOPLEFT point boxes are 1 px/point.
|
||||
px_per_pt = REGIONS_PX_PER_PT if box.get("unit") == "px" else 1.0
|
||||
sx = g["rendered_w"] / (g["page_pt_w"] * px_per_pt)
|
||||
sy = g["rendered_h"] / (g["page_pt_h"] * px_per_pt)
|
||||
scale = 0.5 if box.get("unit") == "px" else 1.0
|
||||
return {
|
||||
"x": round(float(box["x"]) * sx, 2),
|
||||
"y": round(g["page_top"] + float(box["y"]) * sy, 2),
|
||||
"w": round(float(box["w"]) * sx, 2),
|
||||
"h": round(float(box["h"]) * sy, 2),
|
||||
"x": round(float(box["x"]) * scale, 2),
|
||||
"y": round(g["page_top"] + float(box["y"]) * scale, 2),
|
||||
"w": round(float(box["w"]) * scale, 2),
|
||||
"h": round(float(box["h"]) * scale, 2),
|
||||
}
|
||||
if not {"l", "t", "r", "b"}.issubset(box):
|
||||
return None
|
||||
@@ -562,23 +540,6 @@ def _map_first_pass_to_rows(template_id: str, first_pass: Dict[str, Any], pdf_by
|
||||
response_form = _response_form_from_region_type(region.get("region_type"))
|
||||
if response_form:
|
||||
response_areas.append({"id": _ai_id(template_id, "region", page_index, idx), "template_id": template_id, "question_id": first_part_by_page.get(page_index, default_qid), "page": page_index + 1, "bounds": bounds, "kind": "response", "response_form": response_form, "source": "ai", "confirmed": False, "confidence": _safe_confidence(region.get("confidence")), "derivation": region.get("detection_method") or "opencv-response-region"})
|
||||
# Integrity guard: every response_area/boundary question_id must reference an inserted question
|
||||
# (FK exam_response_areas/exam_boundaries -> exam_questions). On papers where band detection yields
|
||||
# few/no questions but opencv/gemma still emit regions, those regions point at the synthetic
|
||||
# default_qid which was never inserted. Ensure that fallback container question exists and reattach
|
||||
# any orphan child rows to it, so persistence can't violate the FK.
|
||||
qid_set = {q["id"] for q in questions}
|
||||
orphans = [r for r in (response_areas + boundaries) if r.get("question_id") not in qid_set]
|
||||
if orphans:
|
||||
if default_qid not in qid_set:
|
||||
questions.insert(0, {"id": default_qid, "template_id": template_id, "label": "Unassigned",
|
||||
"order": 0, "max_marks": 0, "is_container": True, "source": "ai",
|
||||
"confirmed": False, "confidence": 0.5,
|
||||
"derivation": "auto-map-fallback-container"})
|
||||
qid_set.add(default_qid)
|
||||
for r in orphans:
|
||||
r["question_id"] = default_qid
|
||||
|
||||
return {"questions": questions, "response_areas": response_areas, "boundaries": boundaries, "layout": layout}
|
||||
|
||||
|
||||
@@ -634,9 +595,6 @@ def _run_auto_map_job(job_id: str, ctx: ExamContext, template_id: str, pdf_bytes
|
||||
_set_auto_map_status(job_id, {"status": "running", "template_id": template_id})
|
||||
try:
|
||||
rows = _run_auto_map_merge(ctx, template_id, pdf_bytes, source_label)
|
||||
# Project to Neo4j like the born-digital fast path does — otherwise image-only papers (R3's
|
||||
# primary target, routed here because they need OCR) never reach the graph after auto-map.
|
||||
project_template_safe(template_id)
|
||||
_set_auto_map_status(job_id, {"status": "completed", "template_id": template_id, "counts": {k: len(v) for k, v in rows.items()}})
|
||||
except Exception as exc:
|
||||
logger.exception(f"auto-map job failed for template {template_id}: {exc}")
|
||||
|
||||
@@ -718,16 +718,4 @@ def test_auto_map_ocr_returns_job_id_and_status_completes(monkeypatch):
|
||||
body = status.json()
|
||||
assert body["status"] == "completed"
|
||||
assert body["counts"]["questions"] >= 2
|
||||
|
||||
|
||||
def test_auto_map_ocr_path_projects_to_neo4j(monkeypatch, _stub_projection):
|
||||
# Image-only papers route through the async OCR job; regression: that path must project to Neo4j
|
||||
# like the born-digital fast path, or the graph is never built for the primary target.
|
||||
store = _template_with_source()
|
||||
client, store = make_client(store=store)
|
||||
_patch_auto_map(monkeypatch, store, fast=False)
|
||||
resp = client.post("/api/exam/templates/t1/auto-map")
|
||||
assert resp.status_code == 202
|
||||
# the BackgroundTask runs after the response under TestClient
|
||||
assert "t1" in _stub_projection
|
||||
assert body["template"]["layout"]
|
||||
|
||||
Reference in New Issue
Block a user