Compare commits

..
5 Commits
Author SHA1 Message Date
kcarandClaude Opus 4.8 e48dd73fdf exam: capture sections + choice_groups + marks_confidence in extraction_meta (WS-2 R3)
api-ci-deploy / test-build-deploy (push) Has been cancelled
Persist the paper's question-layout signals (A-level Section A/B, EITHER/OR choice
groups, marks confidence) onto exam_templates.extraction_meta so the setup UI can
show them. Data was carried by the contract but dropped at persistence.

Co-Authored-By: Claude Opus 4.8 <[email protected]>
Claude-Session: https://claude.ai/code/session_01GruxHXxfdp4kZCgAMVFgvV
2026-07-04 13:13:45 +00:00
kcarandClaude Opus 4.8 547836e04b exam: persist exam_response_areas.meta on canvas replace-save (WS-2 item 4)
api-ci-deploy / test-build-deploy (push) Has been cancelled
The full-replace save dropped the rich region meta (figure name/description, OMR
geometry). Add meta to ResponseAreaPayload + the replace insert so a named context
figure survives a round-trip. Pairs with the app carrying name/description through.

Co-Authored-By: Claude Opus 4.8 <[email protected]>
Claude-Session: https://claude.ai/code/session_01GruxHXxfdp4kZCgAMVFgvV
2026-07-04 12:13:13 +00:00
kcarandClaude Opus 4.8 df128508a3 exam: persist command_word + preamble from analyse contract v2 (WS-2)
_map_service_contract_to_rows now carries the two v2 fields onto exam_questions
ghost rows (command_word on leaf parts, preamble jsonb). Requires supabase
migration 78; applied to dev .94.

Co-Authored-By: Claude Opus 4.8 <[email protected]>
Claude-Session: https://claude.ai/code/session_01GruxHXxfdp4kZCgAMVFgvV
2026-07-04 11:56:40 +00:00
kcar 81bf44c6cc Merge P3+P4 app wiring
api-ci-deploy / test-build-deploy (push) Has been cancelled
2026-07-03 02:09:17 +00:00
kcarandClaude Opus 4.8 544d858f62 P3+P4 app: persist audit gate (extraction_meta) + digital-text endpoint
_run_service_extract_merge now records extraction_meta {engine, slug, audit
(cover-total reconciliation), counts} on the template (migration 76) — a durable
trust signal the setup UI shows. New GET /templates/{id}/digital-text proxies the
service's /api/replica for the paper (resolved via extraction_meta.slug), returning
the digital-replica markdown. exam_extract.get_replica client added.

Co-Authored-By: Claude Opus 4.8 <[email protected]>
2026-07-03 02:09:17 +00:00
3 changed files with 52 additions and 2 deletions
+13
View File
@@ -40,6 +40,19 @@ def _get(base: str, slug: str, timeout: int = 30) -> Dict[str, Any]:
return r.json() return r.json()
def get_replica(slug: str, timeout: int = 30) -> Dict[str, Any]:
"""The digital-replica markdown for a paper (P4): {slug, title, n_questions, total_marks, markdown,
questions:[{label, marks, markdown}]}. Raises ExtractError if the paper has no replica yet."""
base = service_url()
if not base:
raise ExtractError("EXAM_EXTRACT_URL not configured")
r = requests.get(f"{base}/api/replica/{slug}", timeout=timeout)
if r.status_code == 404:
raise ExtractError(f"no digital replica for {slug}")
r.raise_for_status()
return r.json()
def extract_suggestions(slug: str, pdf_bytes: bytes, *, force: bool = False, def extract_suggestions(slug: str, pdf_bytes: bytes, *, force: bool = False,
poll_timeout: int = 1500, poll_interval: int = 5) -> Dict[str, Any]: poll_timeout: int = 1500, poll_interval: int = 5) -> Dict[str, Any]:
"""POST the paper to the service and poll until the analyse contract is ready. """POST the paper to the service and poll until the analyse contract is ready.
+3
View File
@@ -80,6 +80,9 @@ class ResponseAreaPayload(BaseModel):
] = None ] = None
# Optional Context differentiation (v1 generic; future graph/chart/data_table/diagram/code_block/passage). # Optional Context differentiation (v1 generic; future graph/chart/data_table/diagram/code_block/passage).
context_type: Optional[str] = None context_type: Optional[str] = None
# Rich recognition payload (75-exam-marker-region-meta.sql): figure name/description, OMR geometry, unit…
# Carried on canvas save so a named context figure survives a round-trip.
meta: Optional[Dict[str, Any]] = None
source: Literal["manual", "ai"] = "manual" source: Literal["manual", "ai"] = "manual"
confirmed: bool = True confirmed: bool = True
confidence: Optional[float] = Field(default=None, ge=0, le=1) confidence: Optional[float] = Field(default=None, ge=0, le=1)
+36 -2
View File
@@ -745,6 +745,9 @@ def _map_service_contract_to_rows(template_id: str, contract: Dict[str, Any],
"bounds": _frac_box_to_canvas(q.get("bounds"), q.get("page") or 1, pages), "bounds": _frac_box_to_canvas(q.get("bounds"), q.get("page") or 1, pages),
"page": q.get("page"), "source": "ai", "confirmed": False, "page": q.get("page"), "source": "ai", "confirmed": False,
"confidence": _safe_confidence(q.get("confidence")), "derivation": "extract-service", "confidence": _safe_confidence(q.get("confidence")), "derivation": "extract-service",
# analyse contract v2 (migration 78): command verb + stem prose per part
"command_word": (q.get("command_word") or None) if not q.get("is_container") else None,
"preamble": q.get("preamble") or None,
}) })
for q in questions: # FK safety: de-parent a dangling parent_id for q in questions: # FK safety: de-parent a dangling parent_id
if q["parent_id"] and q["parent_id"] not in q_ids: if q["parent_id"] and q["parent_id"] not in q_ids:
@@ -793,9 +796,21 @@ def _run_service_extract_merge(ctx: ExamContext, template_id: str, pdf_bytes: by
contract = exam_extract.extract_suggestions(slug, pdf_bytes) contract = exam_extract.extract_suggestions(slug, pdf_bytes)
rows = _map_service_contract_to_rows(template_id, contract, pdf_bytes) rows = _map_service_contract_to_rows(template_id, contract, pdf_bytes)
_refresh_ai_rows(ctx, template_id, rows) _refresh_ai_rows(ctx, template_id, rows)
n_pages = (contract.get("meta") or {}).get("n_pages") or (contract.get("meta") or {}).get("pages") meta = contract.get("meta") or {}
# P3: record provenance + the audit cover-reconciliation gate so the setup UI can flag under-reads,
# and store the slug so the digital-text view can resolve this paper's replica.
updates: Dict[str, Any] = {"extraction_meta": {
"engine": "extract-service", "slug": slug, "audit": meta.get("audit") or {},
"counts": {k: len(v) for k, v in rows.items()},
# question-layout signals for the setup UI: paper sections (e.g. A-level Section A/B) + EITHER/OR choices
"sections": meta.get("sections") or [],
"choice_groups": meta.get("choice_groups") or [],
"marks_confidence": meta.get("marks_confidence"),
}}
n_pages = meta.get("n_pages") or meta.get("pages")
if n_pages: if n_pages:
ctx.supabase.table("exam_templates").update({"page_count": n_pages}).eq("id", template_id).execute() updates["page_count"] = n_pages
ctx.supabase.table("exam_templates").update(updates).eq("id", template_id).execute()
return rows return rows
@@ -1034,6 +1049,24 @@ async def auto_map_status(
return body return body
@router.get("/templates/{template_id}/digital-text")
async def template_digital_text(
template_id: str,
ctx: ExamContext = Depends(get_exam_context),
) -> Dict[str, Any]:
"""P4: the digital-replica markdown for this template's paper (stem text, parts, marks, answer-space
placeholders, spec refs). Resolves the paper via the slug the extraction used."""
template = _fetch_template_or_404(ctx, template_id)
_require_source_visibility_or_404(ctx, template)
if not exam_extract.is_enabled():
raise HTTPException(status_code=503, detail="Extraction service not configured")
slug = (template.get("extraction_meta") or {}).get("slug") or _extract_slug(ctx, template)
try:
return exam_extract.get_replica(slug)
except exam_extract.ExtractError as exc:
raise HTTPException(status_code=404, detail=f"No digital text yet — run auto-map first ({exc})")
@router.put("/templates/{template_id}") @router.put("/templates/{template_id}")
async def replace_template( async def replace_template(
template_id: str, template_id: str,
@@ -1114,6 +1147,7 @@ async def replace_template(
"kind": ra.kind, "kind": ra.kind,
"response_form": ra.response_form, "response_form": ra.response_form,
"context_type": ra.context_type, # 73: optional Context differentiation "context_type": ra.context_type, # 73: optional Context differentiation
"meta": ra.meta, # 75: rich recognition payload (name/description/OMR/…)
"source": ra.source, "source": ra.source,
"confirmed": ra.confirmed, "confirmed": ra.confirmed,
"confidence": ra.confidence, "confidence": ra.confidence,