api/run/initialization/init_exam_graph.py
CC Worker 47e45f59e8 FX-6: seed SpecPoint catalogue for all 6 test specs (ASSESSES beyond AQA physics)
Only AQA GCSE Physics 8463 (4.1-4.8) was seeded, so any other board/spec — or
any spec_ref that didn't match those 8 — projected zero (:Part)-[:ASSESSES]->
(:SpecPoint) edges, silently. Seed the full top-level topic catalogue for the 6
current test specs (GCSE + A-level Physics/Chemistry/Biology, 44 SpecPoints) and
loop the Specification/SpecPoint MERGE over all of them (board created once).

Idempotent; deterministic uuid5 keys unchanged. Sub-point granularity remains a
later data task. Run: python3 -c "from run.initialization.init_exam_graph import
init; import json; print(json.dumps(init()))" in the ccapi container.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-02 21:25:23 +00:00

163 lines
8.5 KiB
Python

"""
init_exam_graph.py — Initialise the cc.public.exams Neo4j knowledge graph.
Creates the shared, public exam database, its uniqueness constraints, and seeds the AQA exam board
+ the 6 current test specifications (GCSE & A-level Physics/Chemistry/Biology) with their top-level
topic SpecPoints (44 in total). Idempotent (CREATE DATABASE IF NOT EXISTS / CREATE CONSTRAINT IF NOT
EXISTS / MERGE).
Run inside the ccapi container:
python3 -c "from run.initialization.init_exam_graph import init; import json; print(json.dumps(init()))"
NOTE: only *top-level* topics are seeded (the granularity a teacher plans against). The full sub-point
breakdown (e.g. 4.1.1.1 ...) is a later data-population task (sourceable from the AQA spec PDF via
Docling). Seeding all 6 specs means a template's spec_ref finds a matching SpecPoint so
(:Part)-[:ASSESSES]->(:SpecPoint) fires beyond GCSE Physics; spec_code (e.g. AQA-PHYS-8463) must match
the eb_exams/eb_specifications seed (card S4-3) and the app's deriveSpecCode.
"""
import uuid
from typing import Dict, Any
from modules.database.tools.neo4j_driver_tools import get_driver
EXAM_DB = "cc.public.exams"
NS = uuid.UUID("00000000-0000-0000-0000-00000000e8a1") # stable namespace for deterministic uuids
BOARD = {"code": "AQA", "name": "AQA"}
SPEC = {
"spec_code": "AQA-PHYS-8463",
"exam_board_code": "AQA",
"subject_code": "PHYS",
"award_code": "GCSE",
"title": "AQA GCSE Physics (8463)",
}
# Real AQA GCSE Physics (8463) top-level topics (ref = topic number).
SPEC_POINTS = [
("4.1", "Energy"),
("4.2", "Electricity"),
("4.3", "Particle model of matter"),
("4.4", "Atomic structure"),
("4.5", "Forces"),
("4.6", "Waves"),
("4.7", "Magnetism and electromagnetism"),
("4.8", "Space physics"),
]
# Full AQA catalogue for the current test specs (top-level topics; ref = topic number). Seeding all of
# them means a template's spec_ref finds a matching SpecPoint so (:Part)-[:ASSESSES]->(:SpecPoint) fires
# beyond AQA GCSE Physics. Sub-point granularity (e.g. 4.1.1.1) remains a later data-population task.
SPECIFICATIONS = [
{**SPEC, "topics": SPEC_POINTS},
{"spec_code": "AQA-CHEM-8462", "exam_board_code": "AQA", "subject_code": "CHEM", "award_code": "GCSE",
"title": "AQA GCSE Chemistry (8462)", "topics": [
("4.1", "Atomic structure and the periodic table"), ("4.2", "Bonding, structure, and the properties of matter"),
("4.3", "Quantitative chemistry"), ("4.4", "Chemical changes"), ("4.5", "Energy changes"),
("4.6", "The rate and extent of chemical change"), ("4.7", "Organic chemistry"), ("4.8", "Chemical analysis"),
("4.9", "Chemistry of the atmosphere"), ("4.10", "Using resources")]},
{"spec_code": "AQA-BIOL-8461", "exam_board_code": "AQA", "subject_code": "BIOL", "award_code": "GCSE",
"title": "AQA GCSE Biology (8461)", "topics": [
("4.1", "Cell biology"), ("4.2", "Organisation"), ("4.3", "Infection and response"), ("4.4", "Bioenergetics"),
("4.5", "Homeostasis and response"), ("4.6", "Inheritance, variation and evolution"), ("4.7", "Ecology")]},
{"spec_code": "AQA-PHYS-7408", "exam_board_code": "AQA", "subject_code": "PHYS", "award_code": "A-level",
"title": "AQA A-level Physics (7408)", "topics": [
("3.1", "Measurements and their errors"), ("3.2", "Particles and radiation"), ("3.3", "Waves"),
("3.4", "Mechanics and materials"), ("3.5", "Electricity"), ("3.6", "Further mechanics and thermal physics"),
("3.7", "Fields and their consequences"), ("3.8", "Nuclear physics")]},
{"spec_code": "AQA-CHEM-7405", "exam_board_code": "AQA", "subject_code": "CHEM", "award_code": "A-level",
"title": "AQA A-level Chemistry (7405)", "topics": [
("3.1", "Physical chemistry"), ("3.2", "Inorganic chemistry"), ("3.3", "Organic chemistry")]},
{"spec_code": "AQA-BIOL-7402", "exam_board_code": "AQA", "subject_code": "BIOL", "award_code": "A-level",
"title": "AQA A-level Biology (7402)", "topics": [
("3.1", "Biological molecules"), ("3.2", "Cells"), ("3.3", "Organisms exchange substances with their environment"),
("3.4", "Genetic information, variation and relationships between organisms"),
("3.5", "Energy transfers in and between organisms"), ("3.6", "Organisms respond to changes"),
("3.7", "Genetics, populations, evolution and ecosystems"), ("3.8", "The control of gene expression")]},
]
CONSTRAINTS = [
"CREATE CONSTRAINT exam_board_uid IF NOT EXISTS FOR (n:ExamBoard) REQUIRE n.uuid_string IS UNIQUE",
"CREATE CONSTRAINT spec_uid IF NOT EXISTS FOR (n:Specification) REQUIRE n.uuid_string IS UNIQUE",
"CREATE CONSTRAINT specpoint_uid IF NOT EXISTS FOR (n:SpecPoint) REQUIRE n.uuid_string IS UNIQUE",
"CREATE CONSTRAINT exampaper_uid IF NOT EXISTS FOR (n:ExamPaper) REQUIRE n.uuid_string IS UNIQUE",
"CREATE CONSTRAINT question_uid IF NOT EXISTS FOR (n:Question) REQUIRE n.uuid_string IS UNIQUE",
"CREATE CONSTRAINT part_uid IF NOT EXISTS FOR (n:Part) REQUIRE n.uuid_string IS UNIQUE",
"CREATE CONSTRAINT region_uid IF NOT EXISTS FOR (n:Region) REQUIRE n.uuid_string IS UNIQUE",
"CREATE CONSTRAINT spec_code_unique IF NOT EXISTS FOR (n:Specification) REQUIRE n.spec_code IS UNIQUE",
"CREATE CONSTRAINT exam_code_unique IF NOT EXISTS FOR (n:ExamPaper) REQUIRE n.exam_code IS UNIQUE",
"CREATE CONSTRAINT board_code_unique IF NOT EXISTS FOR (n:ExamBoard) REQUIRE n.code IS UNIQUE",
]
def _uid(*parts: str) -> str:
return str(uuid.uuid5(NS, ":".join(parts)))
def init() -> Dict[str, Any]:
driver = get_driver()
result: Dict[str, Any] = {"db": EXAM_DB, "constraints": 0, "spec_points": 0}
# 1. database
with driver.session(database="system") as s:
s.run(f"CREATE DATABASE `{EXAM_DB}` IF NOT EXISTS").consume()
# wait for availability
import time
for _ in range(30):
with driver.session(database="system") as s:
st = s.run("SHOW DATABASE $n YIELD currentStatus RETURN currentStatus", n=EXAM_DB).single()
if st and st["currentStatus"] == "online":
break
time.sleep(1)
with driver.session(database=EXAM_DB) as s:
# 2. constraints
for c in CONSTRAINTS:
s.run(c).consume()
result["constraints"] += 1
# 3. board (once)
board_uid = _uid("ExamBoard", BOARD["code"])
s.run(
"MERGE (b:ExamBoard {uuid_string:$uid}) "
"SET b.code=$code, b.name=$name, b.node_storage_path=$nsp",
uid=board_uid, code=BOARD["code"], name=BOARD["name"],
nsp=f"{EXAM_DB}/ExamBoard/{BOARD['code']}",
).consume()
# 4. each specification + its top-level spec points (idempotent MERGE)
for spec in SPECIFICATIONS:
spec_uid = _uid("Specification", spec["spec_code"])
s.run(
"MERGE (sp:Specification {uuid_string:$uid}) "
"SET sp.spec_code=$sc, sp.exam_board_code=$ebc, sp.subject_code=$subj, "
" sp.award_code=$award, sp.title=$title, sp.node_storage_path=$nsp "
"WITH sp MATCH (b:ExamBoard {code:$ebc}) MERGE (b)-[:PUBLISHES]->(sp)",
uid=spec_uid, sc=spec["spec_code"], ebc=spec["exam_board_code"],
subj=spec["subject_code"], award=spec["award_code"], title=spec["title"],
nsp=f"{EXAM_DB}/Specification/{spec['spec_code']}",
).consume()
for ref, desc in spec["topics"]:
sp_uid = _uid("SpecPoint", spec["spec_code"], ref)
s.run(
"MERGE (p:SpecPoint {uuid_string:$uid}) "
"SET p.ref=$ref, p.description=$desc, p.spec_code=$sc, "
" p.exam_board_code=$ebc, p.node_storage_path=$nsp "
"WITH p MATCH (s:Specification {spec_code:$sc}) MERGE (s)-[:HAS_SPEC_POINT]->(p)",
uid=sp_uid, ref=ref, desc=desc, sc=spec["spec_code"],
ebc=spec["exam_board_code"], nsp=f"{EXAM_DB}/SpecPoint/{spec['spec_code']}/{ref}",
).consume()
result["spec_points"] += 1
counts = s.run(
"MATCH (b:ExamBoard) WITH count(b) AS boards "
"MATCH (sp:Specification) WITH boards, count(sp) AS specs "
"MATCH (p:SpecPoint) RETURN boards, specs, count(p) AS spec_points"
).single()
result["verify"] = dict(counts) if counts else {}
return result
if __name__ == "__main__":
import json
print(json.dumps(init(), indent=2, default=str))