#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Fail closed on the local Physics source, lesson, arithmetic and asset receipt."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import tempfile
import xml.etree.ElementTree as ET
from decimal import Decimal
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
MANIFEST = ROOT / "MANIFEST.sha256"
DAYS = list(range(1, 11))
STEMS = ("particle-system", "three-paths", "temperature-record",
         "heat-budget", "model-line")
QCAA = "https://www.qcaa.qld.edu.au"


def need(condition: bool, message: str) -> None:
    if not condition:
        raise SystemExit(f"FAIL: {message}")


def read(name: str) -> str:
    return (ROOT / name).read_text(encoding="utf-8")


def sha(path: Path) -> str:
    return hashlib.sha256(path.read_bytes()).hexdigest()


def sections(name: str) -> dict[int, str]:
    content = read(name)
    marks = list(re.finditer(r"^## Day (\d+) · (.+)$", content, re.MULTILINE))
    need([int(m.group(1)) for m in marks] == DAYS,
         f"{name}: expected Days 1–10 exactly in order")
    return {int(m.group(1)): content[m.end(): marks[i + 1].start()
                                  if i + 1 < len(marks) else len(content)]
            for i, m in enumerate(marks)}


def source_snapshot() -> None:
    snap = json.loads(read("source-snapshot.json"))
    phy, qce, bipm, rights = (snap[key] for key in
                              ("physics", "qce_handbook", "bipm", "qcaa_rights"))
    need(snap["checked_on"] == "2026-09-29" and
         snap["jurisdiction"] == "Queensland, Australia" and
         "not a source-byte hash" in snap["verification_boundary"],
         "source date/jurisdiction/hash boundary drift")
    need(phy["title"] == "Physics 2025 v1.3" and
         phy["publication"] == "January 2026" and
         phy["implementation"] == "students completing in 2026 or beyond" and
         phy["landing_url"] == QCAA + "/senior/senior-subjects/syllabuses/sciences/physics" and
         phy["pdf_url"] == QCAA + "/downloads/senior-qce/syllabuses/snr_physics_25_syll.pdf" and
         phy["unit"] == "Unit 1: Thermal, nuclear and electrical physics" and
         phy["topic"] == "Topic 1: Heating processes" and
         phy["topic_notional_hours"] == 15 and
         phy["printed_pages"] == {"course_design": 5, "assessment_autonomy": 6,
                                  "unit_overview": 19, "topic_science_understanding": 20,
                                  "topic_inquiry": 21} and
         phy["pdf_page_offset"] == 2 and
         len(phy["sampled_rows"]) == 7 and len(phy["explicit_deferrals"]) == 6,
         "official Physics page/version/topic/row-location drift")
    need(qce["url"] == QCAA + "/senior/certificates-and-qualifications/qce-qcia-handbook/4-qld-curriculum/4.1-syllabuses" and
         qce["general_unit_notional_hours"] == 55 and
         bipm["url"] == "https://www.bipm.org/documents/d/guest/si-brochure-9-2_01" and
         rights["url"] == QCAA + "/copyright", "other primary-source drift")
    crosswalk, ledger = read("CURRICULUM-CROSSWALK.md"), read("SOURCE-AND-RIGHTS.md")
    for marker in ("Physics 2025 v1.3", "29 September 2026", "printed pp. 20–21",
                   "15 notional hours", "55 notional hours", "250 minutes",
                   "school decisions", "not secure exams", "not claimed"):
        need(marker.lower() in crosswalk.lower(), f"crosswalk scope marker missing: {marker}")
    for source in (phy["pdf_url"], phy["landing_url"]):
        need(source in crosswalk and source in ledger, f"Physics authority absent: {source}")
    need(all(url in ledger for url in (qce["url"], bipm["url"], rights["url"])) and
         "not redistributed" in ledger.lower(), "primary-source rights/boundary absent")
    need(re.search(r"\bAC9[A-Z0-9]{5,}\b", crosswalk) is None,
         "invented senior AC9 content code")


def lesson_routes() -> dict[int, str]:
    sources, lessons, learners = (sections(name) for name in
                                  ("SOURCE-CARDS.md", "LESSONS.md", "LEARNER.md"))
    for day in DAYS:
        need("**Public" in sources[day] and "**Desk note:" in sources[day] and
             "**Question:**" in sources[day], f"Day {day}: source/desk/question absent")
        need(all(marker in lessons[day] for marker in
                 ("**0–3", "**3–7", "**7–12", "**12–20", "**20–23", "**23–25")),
             f"Day {day}: distinct timed 25-minute phases absent")
        routes = re.findall(r"^- \*\*([ABC]) · ([^*]+):\*\* (.+)$",
                            learners[day], re.MULTILINE)
        need([route[0] for route in routes] == ["A", "B", "C"] and
             all(len(route[2]) > 95 for route in routes) and
             learners[day].count("**Extra 1 / paper home:**") == 1 and
             learners[day].count("**Extra 2 / paper home:**") == 1,
             f"Day {day}: 3 substantive routes or 2 no-purchase extras absent")
    intro = read("LEARNER.md").split("## Day 1")[0].lower()
    need(all(marker in intro for marker in
             ("same three things", "not fixed learning-style labels", "read-aloud")),
         "equivalent-target/access boundary absent")
    need("Do **not** ask learners to perform live heating" in read("README.md") and
         "No learner should heat" in read("SOURCE-CARDS.md"),
         "paper-only safety boundary absent")
    return sources


def checks() -> None:
    page, key = read("STUDENT-CHECKS.md"), read("teacher/KEY-AND-NEXT.md")
    need(re.findall(r"^## Check ([AB]) · after Day (\d+) ·", page, re.MULTILINE) ==
         [("A", "5"), ("B", "10")], "two fresh formative checks absent")
    need("## Check A worked response" in key and "## Check B worked response" in key,
         "separate fresh-check worked key absent")
    for doc in (page, key):
        need("publicly accessible by url" in doc.lower() and
             "not secure exams" in doc.lower(), "public check/key boundary absent")
    practice = "\n".join(read(name) for name in
                         ("SOURCE-CARDS.md", "LESSONS.md", "LEARNER.md"))
    for distinct in ("Light-box route map", "Gallery-board model line", "0.40 kg single-phase sample"):
        need(distinct.lower() not in practice.lower(), f"fresh check leaked into practice: {distinct}")
    for marker in ("evacuated gap", "solid bracket", "outer air-filled housing",
                   "400 J K⁻¹", "1,000 J kg⁻¹ K⁻¹"):
        need(marker in key, f"worked check target missing: {marker}")


def arithmetic(sources: dict[int, str]) -> None:
    d = Decimal
    key = read("teacher/KEY-AND-NEXT.md")
    rows = {int(day): result for day, result, _ in
            re.findall(r"^\| (\d\d) \| (.+) \| (.+) \|$", key, re.MULTILINE)}
    need(sorted(rows) == DAYS, "ten daily worked-key rows absent")
    calculations = {
        6: (d("22") + 273, d("-5") + 273, (d("-5") + 273) - (d("22") + 273)),
        7: (d("27.0") - d("18.0"), d("0.1") + d("0.1"),
            (d("0.1") + d("0.1")) / (d("27.0") - d("18.0")) * 100),
        8: (d("0.50") * d("2000") * 4, d("1.00") * d("2000") * 4),
        9: (d("3000") / (d("0.50") * d("1000")),
            d("3000") / (d("0.50") * d("2000"))),
        10: ((d("3000") - d("1000")) / (d("3") - d("1")),
             ((d("3000") - d("1000")) / (d("3") - d("1"))) / d("0.25")),
    }
    expected = {6: ("295", "268", "-27"), 7: ("9.0", "0.2"),
                8: ("4000", "8000"), 9: ("6", "3"), 10: ("1000", "4000")}
    source_tokens = {
        6: ("22 °C", "−5 °C", "T(K) = T(°C) + 273"),
        7: ("18.0 °C", "27.0 °C", "±0.1 °C"),
        8: ("2,000 J kg⁻¹ K⁻¹", "0.50 kg", "1.00 kg", "ΔT = 4 K"),
        9: ("0.50 kg", "3,000 J", "1,000 J kg⁻¹ K⁻¹", "2,000 J kg⁻¹ K⁻¹"),
        10: ("0.25 kg", "(1, 1,000)", "(3, 3,000)"),
    }
    for day in range(6, 11):
        need(all(token in sources[day] for token in source_tokens[day]),
             f"Day {day}: source model givens drift")
        for index, value in enumerate(expected[day]):
            need(calculations[day][index] == d(value),
                 f"Day {day}: exact arithmetic drift at value {index}")
            normal = rows[day].replace(",", "").replace("−", "-")
            need(re.search(rf"(?<!\d){re.escape(value)}(?!\d)", normal) is not None,
                 f"Day {day}: worked key missing {value}")
    need(d("2.1") < calculations[7][2] < d("2.3") and "2.2%" in rows[7],
         "Day 7 percentage rounding drift")
    slope = (d("2400") - d("800")) / (d("6") - d("2"))
    need(slope == d("400") and slope / d("0.40") == d("1000"),
         "fresh Check B model slope/c drift")
    page = read("STUDENT-CHECKS.md")
    need(all(token in page for token in ("0.40 kg", "(2, 800)", "(6, 2,400)")),
         "fresh Check B givens drift")


def assets() -> None:
    text_route = read("print/TEXT-ALTERNATIVES.md")
    need(text_route.count("**Tactile route:**") == 5 and all(
        title in text_route for title in
        ("## Particle-and-system mat", "## Three-paths mat", "## Temperature-record mat",
         "## Heat-budget mat", "## Model-line mat")),
        "five complete text/tactile aids absent")
    ns = {"svg": "http://www.w3.org/2000/svg"}
    for stem in STEMS:
        svg = ROOT / "print" / f"{stem}.svg"
        pdf = svg.with_suffix(".pdf")
        root = ET.parse(svg).getroot()
        need(root.attrib.get("width") == "210mm" and root.attrib.get("height") == "297mm" and
             root.attrib.get("viewBox") == "0 0 210 297" and
             root.attrib.get("role") == "img" and
             root.attrib.get("aria-labelledby") == "title desc" and
             root.find("svg:title", ns) is not None and
             root.find("svg:desc", ns) is not None,
             f"{stem}: SVG A4/access labels absent")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        words = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        fonts = subprocess.check_output(["pdffonts", str(pdf)], text=True)
        need("(A4)" in info and re.search(r"Pages:\s+1\b", info) is not None and
             "SubjectNest" in words and "CC BY 4.0" in words and len(words) > 170 and
             "DejaVuSans" in fonts, f"{stem}: PDF page/text/font drift")
    with tempfile.TemporaryDirectory(prefix="subjectnest-y11-physics-") as temp:
        subprocess.run([sys.executable, str(ROOT / "print/generate_print.py"),
                        "--output-dir", temp], capture_output=True, check=True, text=True)
        for stem in STEMS:
            for extension in ("svg", "pdf"):
                name = f"{stem}.{extension}"
                need(sha(Path(temp) / name) == sha(ROOT / "print" / name),
                     f"{name}: print generation drift")
    html = read("interactive/heat-budget-lab.html")
    need(all(marker in html for marker in
             ('id="heat-form"', 'id="result"', 'aria-live="polite"',
              'id="prediction"', 'id="variation"', 'const first = mass * capacity * change')),
         "offline lab prediction/model/access markers absent")
    need(not any(marker in html for marker in
                 ("fetch(", "XMLHttpRequest", "localStorage", "sessionStorage",
                  "sendBeacon", "<script src=", '<link rel="stylesheet"')),
         "offline lab external network/storage dependency")
    paper = read("interactive/TEXT-ROUTE.md")
    need(all(marker in paper for marker in
             ("4,000 J", "8,000 J", "3,000 J", "1,500 J", "**Tactile route:**")),
         "interactive paper equivalent incomplete")


def slug(heading: str) -> str:
    heading = re.sub(r"<[^>]+>", "", heading)
    return re.sub(r"[^a-z0-9_-]", "", heading.lower().strip().replace(" ", "-"))


def links_and_rights() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        content = path.read_text(encoding="utf-8")
        need("CC BY 4.0" in content, f"{path.name}: original-rights notice absent")
        for raw in re.findall(r"\[[^]]+\]\(([^)]+)\)", content):
            if raw.startswith(("https://", "http://", "mailto:")):
                continue
            relative, _, anchor = unquote(raw).partition("#")
            target = (path.parent / relative).resolve() if relative else path
            need(target.exists(), f"{path.name}: broken local link {raw}")
            if anchor and target.suffix.lower() == ".md":
                headings = [slug(m.group(1)) for m in re.finditer(
                    r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)]
                need(anchor in headings, f"{path.name}: broken fragment {raw}")
            count += 1
    html = read("interactive/heat-budget-lab.html")
    for raw in re.findall(r'href="([^"]+)"', html):
        if raw.startswith(("https://", "http://", "#")):
            continue
        relative, _, anchor = raw.partition("#")
        target = (ROOT / "interactive" / relative).resolve()
        need(target.exists(), f"HTML broken local link {raw}")
        if anchor and target.suffix.lower() == ".md":
            headings = [slug(m.group(1)) for m in re.finditer(
                r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)]
            need(anchor in headings, f"HTML broken fragment {raw}")
        count += 1
    return count


def files() -> list[Path]:
    return sorted((path for path in ROOT.rglob("*") if path.is_file() and
                   path != MANIFEST and "__pycache__" not in path.parts and
                   ".ruff_cache" not in path.parts),
                  key=lambda path: path.relative_to(ROOT).as_posix())


def manifest() -> str:
    return "".join(f"{sha(path)}  {path.relative_to(ROOT).as_posix()}\n" for path in files())


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    source_snapshot()
    sources = lesson_routes()
    checks()
    arithmetic(sources)
    assets()
    links = links_and_rights()
    expected = manifest()
    if args.write_manifest:
        MANIFEST.write_text(expected, encoding="utf-8")
    else:
        need(MANIFEST.read_text(encoding="utf-8") == expected, "SHA manifest drift")
    print(f"PASS: ten 25-minute scripts, 10 source cases, 30 same-target routes, "
          f"20 optional contexts, 2 public fresh checks/keys, 5 A4 SVG/PDF/text aids, "
          f"offline/paper lab, {links} local links, {len(files())} hashed files")


if __name__ == "__main__":
    main()
