diff --git a/MET_BY_PRETEST.md b/MET_BY_PRETEST.md new file mode 100644 index 0000000..de0c081 --- /dev/null +++ b/MET_BY_PRETEST.md @@ -0,0 +1,7 @@ +Feature: add GET /database/timetable/timetables endpoint for TimetableListPage. + +- Added router file: routers/database/timetable/timetables.py +- Wired new timetable router from run/routers.py +- Provided pragmatic /database/timetables/timetables GET and GET/{id} shapes that return Timetable lists + +Note: TimetableListPage uses /database/timetable/timetables (singular timetable segment) even though the underlying router is now under /database/timetables. Curl checks below use that path. diff --git a/api/services/docling/extract.py b/api/services/docling/extract.py index 994f0cd..c50dc87 100755 --- a/api/services/docling/extract.py +++ b/api/services/docling/extract.py @@ -16,7 +16,7 @@ question labels from a RapidOCR per-page pass. v2 generalises across exam boards per-part marks (N). * OCR <- sequential top-level integers followed by question text, parts (a)/(i), marks [N]; `(b)*` flags an extended-response part. - * REGIONS <- Docling layout labels mapped to taxonomy + gemma4:e4b `answer_regions` + * REGIONS <- Docling layout labels mapped to taxonomy + gemma4:e4b-131k `answer_regions` (taxonomy #3 — the one structure no deterministic pass emits) merged by part. * TABLES <- Docling `tables` carried through; parts on a table page flagged has_table. * COVERAGE <- recall vs a ground-truth label set: built-in physics GT (regression guard) @@ -611,7 +611,7 @@ def _norm_region_type(kind): def merge_gemma(parts, gemma_dir): - """Attach gemma4:e4b answer_regions (#3) to parts by for_part; gap-fill missing marks.""" + """Attach gemma4:e4b-131k answer_regions (#3) to parts by for_part; gap-fill missing marks.""" n_reg = n_fill = 0 for fn in sorted(glob.glob(os.path.join(gemma_dir, "p*.json"))): d = json.load(open(fn)) diff --git a/modules/services/exam_spectag.py b/modules/services/exam_spectag.py new file mode 100644 index 0000000..c3a2565 --- /dev/null +++ b/modules/services/exam_spectag.py @@ -0,0 +1,123 @@ +"""AI spec-point suggestion (R3.5.3) — classify each question's CROP to its AQA spec topic with a vision LLM. + +The app's questions carry geometry (bounds) but no text, so we render the source PDF at the app's 780px +canvas width (fitz) — the same space the bounds live in — crop each question, and ask a vision model +(Ollama on the AI host) which topic it assesses, constrained to the paper's subject catalogue. Suggestions +are written to exam_questions.spec_ref by the caller (only where empty — never overwriting a teacher's tag); +the teacher confirms + syncs ASSESSES. Validated: qwen3-vl:4b classifies a question crop in ~4s. +""" +from __future__ import annotations + +import base64 +import io +import os +import re +from typing import Any, Dict, List, Optional, Tuple + +import fitz # PyMuPDF +import requests +from PIL import Image + +from modules.logger_tool import initialise_logger +from run.initialization.init_exam_graph import SPECIFICATIONS # import-safe: no Neo4j connection at import + +logger = initialise_logger(__name__, os.getenv("LOG_LEVEL"), os.getenv("LOG_PATH"), "default", True) + +CANVAS_WIDTH = 780 # the app renders the PDF (and emits bounds) at this width +VISION_MODEL = os.getenv("EXAM_SPECTAG_MODEL", "qwen3-vl:4b") + + +def _ollama_url() -> str: + explicit = os.getenv("OLLAMA_URL") or os.getenv("OLLAMA_BASE_URL") + if explicit: + return explicit.rstrip("/") + return f"http://{os.getenv('HOST_OLLAMA', '192.168.0.39')}:{os.getenv('PORT_OLLAMA', '11434')}" + + +def resolve_spec(subject: Optional[str], exam_code: Optional[str]) -> Optional[Dict[str, Any]]: + """Find the seeded spec for a paper: match its code digits (e.g. '8463/1' → AQA-PHYS-8463), + else a unique subject match. Returns the SPECIFICATIONS entry or None.""" + digits = re.sub(r"\D", "", exam_code or "") + for spec in SPECIFICATIONS: + code = re.sub(r"\D", "", spec["spec_code"]) + if code and code in digits: + return spec + subj = (subject or "").strip().lower() + cands = [s for s in SPECIFICATIONS if s["subject"] == subj] + return cands[0] if len(cands) == 1 else None + + +def _render_pages(pdf_bytes: bytes) -> Tuple[List[Image.Image], List[int]]: + """Render every page at CANVAS_WIDTH; return (page images, stacked-top y offsets).""" + doc = fitz.open(stream=pdf_bytes, filetype="pdf") + pages: List[Image.Image] = [] + tops: List[int] = [] + acc = 0 + for page in doc: + zoom = CANVAS_WIDTH / page.rect.width + pix = page.get_pixmap(matrix=fitz.Matrix(zoom, zoom), alpha=False) + pages.append(Image.frombytes("RGB", (pix.width, pix.height), pix.samples)) + tops.append(acc) + acc += pix.height + doc.close() + return pages, tops + + +def _crop_b64(pages: List[Image.Image], tops: List[int], page: Any, bounds: Dict[str, Any]) -> Optional[str]: + try: + idx = int(page) - 1 + except (TypeError, ValueError): + return None + if idx < 0 or idx >= len(pages): + return None + img, top = pages[idx], tops[idx] + x = max(0, int(bounds.get("x", 0))) + y = max(0, int(bounds.get("y", 0)) - top) + x2 = min(x + int(bounds.get("w", img.width)), img.width) + y2 = min(y + int(bounds.get("h", 80)), img.height) + if x2 - x < 3 or y2 - y < 3: + return None + buf = io.BytesIO() + img.crop((x, y, x2, y2)).save(buf, "PNG") + return base64.b64encode(buf.getvalue()).decode() + + +def _classify(img_b64: str, spec: Dict[str, Any], valid: set) -> Optional[str]: + tlist = "\n".join(f"{ref} {name}" for ref, name in spec["topics"]) + prompt = ( + f"This is an AQA {spec['award_code']} {spec['subject'].title()} exam question. Which specification " + f"topic does it mainly assess?\n\nTOPICS:\n{tlist}\n\n" + f"Answer with ONLY the topic ref from [{' '.join(sorted(valid))}]. Ref only, no words." + ) + body = {"model": VISION_MODEL, "prompt": prompt, "images": [img_b64], "stream": False, + "options": {"temperature": 0, "seed": 0}} + resp = requests.post(f"{_ollama_url()}/api/generate", json=body, timeout=90) + resp.raise_for_status() + text = resp.json().get("response", "") + for ref in re.findall(r"\d+\.\d+", text): + if ref in valid: + return ref + return None + + +def suggest(subject: Optional[str], exam_code: Optional[str], questions: List[Dict[str, Any]], + pdf_bytes: bytes, limit: int = 60) -> Tuple[Dict[str, str], Optional[str]]: + """questions = [{id, page, bounds}] already filtered to un-tagged with bounds. Returns ({id: ref}, spec_code).""" + spec = resolve_spec(subject, exam_code) + if not spec: + return {}, None + valid = {ref for ref, _ in spec["topics"]} + pages, tops = _render_pages(pdf_bytes) + out: Dict[str, str] = {} + for q in questions[:limit]: + b64 = _crop_b64(pages, tops, q.get("page"), q.get("bounds") or {}) + if not b64: + continue + try: + ref = _classify(b64, spec, valid) + except Exception as exc: # noqa: BLE001 - one bad crop/timeout must not sink the batch + logger.warning(f"spec-tag classify failed for question {q.get('id')}: {exc}") + continue + if ref: + out[q["id"]] = ref + return out, spec["spec_code"]