spike: preserve exam spec-tagging prototype
This commit is contained in:
@@ -0,0 +1,7 @@
|
|||||||
|
Feature: add GET /database/timetable/timetables endpoint for TimetableListPage.
|
||||||
|
|
||||||
|
- Added router file: routers/database/timetable/timetables.py
|
||||||
|
- Wired new timetable router from run/routers.py
|
||||||
|
- Provided pragmatic /database/timetables/timetables GET and GET/{id} shapes that return Timetable lists
|
||||||
|
|
||||||
|
Note: TimetableListPage uses /database/timetable/timetables (singular timetable segment) even though the underlying router is now under /database/timetables. Curl checks below use that path.
|
||||||
@@ -16,7 +16,7 @@ question labels from a RapidOCR per-page pass. v2 generalises across exam boards
|
|||||||
per-part marks (N).
|
per-part marks (N).
|
||||||
* OCR <- sequential top-level integers followed by question text, parts (a)/(i),
|
* OCR <- sequential top-level integers followed by question text, parts (a)/(i),
|
||||||
marks [N]; `(b)*` flags an extended-response part.
|
marks [N]; `(b)*` flags an extended-response part.
|
||||||
* REGIONS <- Docling layout labels mapped to taxonomy + gemma4:e4b `answer_regions`
|
* REGIONS <- Docling layout labels mapped to taxonomy + gemma4:e4b-131k `answer_regions`
|
||||||
(taxonomy #3 — the one structure no deterministic pass emits) merged by part.
|
(taxonomy #3 — the one structure no deterministic pass emits) merged by part.
|
||||||
* TABLES <- Docling `tables` carried through; parts on a table page flagged has_table.
|
* TABLES <- Docling `tables` carried through; parts on a table page flagged has_table.
|
||||||
* COVERAGE <- recall vs a ground-truth label set: built-in physics GT (regression guard)
|
* COVERAGE <- recall vs a ground-truth label set: built-in physics GT (regression guard)
|
||||||
@@ -611,7 +611,7 @@ def _norm_region_type(kind):
|
|||||||
|
|
||||||
|
|
||||||
def merge_gemma(parts, gemma_dir):
|
def merge_gemma(parts, gemma_dir):
|
||||||
"""Attach gemma4:e4b answer_regions (#3) to parts by for_part; gap-fill missing marks."""
|
"""Attach gemma4:e4b-131k answer_regions (#3) to parts by for_part; gap-fill missing marks."""
|
||||||
n_reg = n_fill = 0
|
n_reg = n_fill = 0
|
||||||
for fn in sorted(glob.glob(os.path.join(gemma_dir, "p*.json"))):
|
for fn in sorted(glob.glob(os.path.join(gemma_dir, "p*.json"))):
|
||||||
d = json.load(open(fn))
|
d = json.load(open(fn))
|
||||||
|
|||||||
@@ -0,0 +1,123 @@
|
|||||||
|
"""AI spec-point suggestion (R3.5.3) — classify each question's CROP to its AQA spec topic with a vision LLM.
|
||||||
|
|
||||||
|
The app's questions carry geometry (bounds) but no text, so we render the source PDF at the app's 780px
|
||||||
|
canvas width (fitz) — the same space the bounds live in — crop each question, and ask a vision model
|
||||||
|
(Ollama on the AI host) which topic it assesses, constrained to the paper's subject catalogue. Suggestions
|
||||||
|
are written to exam_questions.spec_ref by the caller (only where empty — never overwriting a teacher's tag);
|
||||||
|
the teacher confirms + syncs ASSESSES. Validated: qwen3-vl:4b classifies a question crop in ~4s.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import base64
|
||||||
|
import io
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
from typing import Any, Dict, List, Optional, Tuple
|
||||||
|
|
||||||
|
import fitz # PyMuPDF
|
||||||
|
import requests
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
|
from modules.logger_tool import initialise_logger
|
||||||
|
from run.initialization.init_exam_graph import SPECIFICATIONS # import-safe: no Neo4j connection at import
|
||||||
|
|
||||||
|
logger = initialise_logger(__name__, os.getenv("LOG_LEVEL"), os.getenv("LOG_PATH"), "default", True)
|
||||||
|
|
||||||
|
CANVAS_WIDTH = 780 # the app renders the PDF (and emits bounds) at this width
|
||||||
|
VISION_MODEL = os.getenv("EXAM_SPECTAG_MODEL", "qwen3-vl:4b")
|
||||||
|
|
||||||
|
|
||||||
|
def _ollama_url() -> str:
|
||||||
|
explicit = os.getenv("OLLAMA_URL") or os.getenv("OLLAMA_BASE_URL")
|
||||||
|
if explicit:
|
||||||
|
return explicit.rstrip("/")
|
||||||
|
return f"http://{os.getenv('HOST_OLLAMA', '192.168.0.39')}:{os.getenv('PORT_OLLAMA', '11434')}"
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_spec(subject: Optional[str], exam_code: Optional[str]) -> Optional[Dict[str, Any]]:
|
||||||
|
"""Find the seeded spec for a paper: match its code digits (e.g. '8463/1' → AQA-PHYS-8463),
|
||||||
|
else a unique subject match. Returns the SPECIFICATIONS entry or None."""
|
||||||
|
digits = re.sub(r"\D", "", exam_code or "")
|
||||||
|
for spec in SPECIFICATIONS:
|
||||||
|
code = re.sub(r"\D", "", spec["spec_code"])
|
||||||
|
if code and code in digits:
|
||||||
|
return spec
|
||||||
|
subj = (subject or "").strip().lower()
|
||||||
|
cands = [s for s in SPECIFICATIONS if s["subject"] == subj]
|
||||||
|
return cands[0] if len(cands) == 1 else None
|
||||||
|
|
||||||
|
|
||||||
|
def _render_pages(pdf_bytes: bytes) -> Tuple[List[Image.Image], List[int]]:
|
||||||
|
"""Render every page at CANVAS_WIDTH; return (page images, stacked-top y offsets)."""
|
||||||
|
doc = fitz.open(stream=pdf_bytes, filetype="pdf")
|
||||||
|
pages: List[Image.Image] = []
|
||||||
|
tops: List[int] = []
|
||||||
|
acc = 0
|
||||||
|
for page in doc:
|
||||||
|
zoom = CANVAS_WIDTH / page.rect.width
|
||||||
|
pix = page.get_pixmap(matrix=fitz.Matrix(zoom, zoom), alpha=False)
|
||||||
|
pages.append(Image.frombytes("RGB", (pix.width, pix.height), pix.samples))
|
||||||
|
tops.append(acc)
|
||||||
|
acc += pix.height
|
||||||
|
doc.close()
|
||||||
|
return pages, tops
|
||||||
|
|
||||||
|
|
||||||
|
def _crop_b64(pages: List[Image.Image], tops: List[int], page: Any, bounds: Dict[str, Any]) -> Optional[str]:
|
||||||
|
try:
|
||||||
|
idx = int(page) - 1
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
if idx < 0 or idx >= len(pages):
|
||||||
|
return None
|
||||||
|
img, top = pages[idx], tops[idx]
|
||||||
|
x = max(0, int(bounds.get("x", 0)))
|
||||||
|
y = max(0, int(bounds.get("y", 0)) - top)
|
||||||
|
x2 = min(x + int(bounds.get("w", img.width)), img.width)
|
||||||
|
y2 = min(y + int(bounds.get("h", 80)), img.height)
|
||||||
|
if x2 - x < 3 or y2 - y < 3:
|
||||||
|
return None
|
||||||
|
buf = io.BytesIO()
|
||||||
|
img.crop((x, y, x2, y2)).save(buf, "PNG")
|
||||||
|
return base64.b64encode(buf.getvalue()).decode()
|
||||||
|
|
||||||
|
|
||||||
|
def _classify(img_b64: str, spec: Dict[str, Any], valid: set) -> Optional[str]:
|
||||||
|
tlist = "\n".join(f"{ref} {name}" for ref, name in spec["topics"])
|
||||||
|
prompt = (
|
||||||
|
f"This is an AQA {spec['award_code']} {spec['subject'].title()} exam question. Which specification "
|
||||||
|
f"topic does it mainly assess?\n\nTOPICS:\n{tlist}\n\n"
|
||||||
|
f"Answer with ONLY the topic ref from [{' '.join(sorted(valid))}]. Ref only, no words."
|
||||||
|
)
|
||||||
|
body = {"model": VISION_MODEL, "prompt": prompt, "images": [img_b64], "stream": False,
|
||||||
|
"options": {"temperature": 0, "seed": 0}}
|
||||||
|
resp = requests.post(f"{_ollama_url()}/api/generate", json=body, timeout=90)
|
||||||
|
resp.raise_for_status()
|
||||||
|
text = resp.json().get("response", "")
|
||||||
|
for ref in re.findall(r"\d+\.\d+", text):
|
||||||
|
if ref in valid:
|
||||||
|
return ref
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def suggest(subject: Optional[str], exam_code: Optional[str], questions: List[Dict[str, Any]],
|
||||||
|
pdf_bytes: bytes, limit: int = 60) -> Tuple[Dict[str, str], Optional[str]]:
|
||||||
|
"""questions = [{id, page, bounds}] already filtered to un-tagged with bounds. Returns ({id: ref}, spec_code)."""
|
||||||
|
spec = resolve_spec(subject, exam_code)
|
||||||
|
if not spec:
|
||||||
|
return {}, None
|
||||||
|
valid = {ref for ref, _ in spec["topics"]}
|
||||||
|
pages, tops = _render_pages(pdf_bytes)
|
||||||
|
out: Dict[str, str] = {}
|
||||||
|
for q in questions[:limit]:
|
||||||
|
b64 = _crop_b64(pages, tops, q.get("page"), q.get("bounds") or {})
|
||||||
|
if not b64:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
ref = _classify(b64, spec, valid)
|
||||||
|
except Exception as exc: # noqa: BLE001 - one bad crop/timeout must not sink the batch
|
||||||
|
logger.warning(f"spec-tag classify failed for question {q.get('id')}: {exc}")
|
||||||
|
continue
|
||||||
|
if ref:
|
||||||
|
out[q["id"]] = ref
|
||||||
|
return out, spec["spec_code"]
|
||||||
Reference in New Issue
Block a user