#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Fail closed on the Year 11 Health source, count, privacy and print receipt."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import tempfile
import xml.etree.ElementTree as ET
from decimal import Decimal
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
MANIFEST = ROOT / "MANIFEST.sha256"
DAYS = list(range(1, 11))
STEMS = ("lens-language", "resource-access", "message-evidence",
         "dimension-map", "issue-question")
QCAA = "https://www.qcaa.qld.edu.au"


def need(condition: bool, message: str) -> None:
    if not condition:
        raise SystemExit(f"FAIL: {message}")


def read(name: str) -> str:
    return (ROOT / name).read_text(encoding="utf-8")


def sha(path: Path) -> str:
    return hashlib.sha256(path.read_bytes()).hexdigest()


def sections(name: str) -> dict[int, str]:
    data = read(name)
    marks = list(re.finditer(r"^## Day (\d+) · (.+)$", data, re.MULTILINE))
    need([int(m.group(1)) for m in marks] == DAYS,
         f"{name}: expected Days 1–10 exactly in order")
    return {int(m.group(1)): data[m.end(): marks[i + 1].start()
                                  if i + 1 < len(marks) else len(data)]
            for i, m in enumerate(marks)}


def source_snapshot() -> None:
    snap = json.loads(read("source-snapshot.json"))
    health, qce, rights = (snap[key] for key in
                           ("health", "qce_handbook", "qcaa_rights"))
    need(snap["checked_on"] == "2026-09-29" and
         snap["jurisdiction"] == "Queensland, Australia" and
         "not redistributed or hashed" in snap["verification_boundary"],
         "source date/jurisdiction/hash boundary drift")
    need(health["landing_url"] == QCAA + "/senior/senior-subjects/syllabuses/health-physical-education/health" and
         health["pdf_url"] == QCAA + "/downloads/senior-qce/syllabuses/snr_health_25_syll.pdf" and
         health["title"] == "Health 2025 v1.3" and
         health["publication"] == "January 2026" and
         health["implementation"] == "students completing in 2026 or beyond" and
         health["unit"] == "Unit 1: Resilience as a personal health resource" and
         health["opening_stage"] == "Stage 1: Define and understand resilience as a personal health resource" and
         health["no_numbered_topic_1"] is True and
         health["printed_pages"] == {"course_design": 5, "assessment_autonomy": 6,
                                     "health_inquiry_model": 8, "unit_overview": 12,
                                     "unit_objectives": 13, "stage_1_start": 14,
                                     "stage_1_middle": 15, "stage_1_end": 16,
                                     "stage_2_start": 17} and
         health["pdf_page_offset"] == 2 and
         len(health["sampled_opportunities"]) == 7 and len(health["deferred"]) == 7,
         "Health version/unit/Stage 1 location drift")
    need(qce["url"] == QCAA + "/senior/certificates-and-qualifications/qce-qcia-handbook/4-qld-curriculum/4.1-syllabuses" and
         qce["general_unit_notional_hours"] == 55 and
         rights["url"] == QCAA + "/copyright", "QCE/rights authority drift")
    crosswalk, ledger = read("CURRICULUM-CROSSWALK.md"), read("SOURCE-AND-RIGHTS.md")
    for marker in ("Health 2025 v1.3", "29 September 2026", "printed pp. 14–16",
                   "no numbered", "250 minutes", "55 notional hours",
                   "school decisions", "not secure exams", "not claimed"):
        need(marker.lower() in crosswalk.lower(), f"crosswalk scope missing: {marker}")
    for source in (health["pdf_url"], health["landing_url"]):
        need(source in crosswalk and source in ledger,
             f"official Health source absent: {source}")
    need(qce["url"] in crosswalk and qce["url"] in ledger and
         rights["url"] in ledger and "linked, not redistributed" in ledger.lower(),
         "official source/rights boundary missing")
    need(re.search(r"\bAC9[A-Z0-9]{5,}\b", crosswalk) is None,
         "invented senior AC9 content code")


def lesson_routes() -> dict[int, str]:
    sources, lessons, learners = (sections(name) for name in
                                  ("SOURCE-CARDS.md", "LESSONS.md", "LEARNER.md"))
    for day in DAYS:
        need("**Public message:**" in sources[day] and
             "**Desk note:**" in sources[day] and
             "**Inquiry:**" in sources[day], f"Day {day}: source pair/question absent")
        need(all(marker in lessons[day] for marker in
                 ("**0–3", "**3–7", "**7–12", "**12–20", "**20–23", "**23–25")),
             f"Day {day}: timed 25-minute script incomplete")
        routes = re.findall(r"^- \*\*([ABC]) · ([^*]+):\*\* (.+)$",
                            learners[day], re.MULTILINE)
        need([route[0] for route in routes] == ["A", "B", "C"] and
             all(len(route[2]) > 90 for route in routes) and
             learners[day].count("**Extra 1 / paper home:**") == 1 and
             learners[day].count("**Extra 2 / paper home:**") == 1,
             f"Day {day}: 3 substantial routes/2 extras missing")
    intro = read("LEARNER.md").split("## Day 1")[0].lower()
    need(all(marker in intro for marker in
             ("same three things", "not fixed learning-style labels", "read-aloud",
              "do not disclose")), "common evidence/access/privacy target absent")
    need("no personal health disclosure" in read("LESSONS.md").lower() and
         "do not collect personal disclosures" in read("SOURCE-CARDS.md").lower(),
         "no-disclosure teacher/source boundary absent")
    return sources


def checks() -> None:
    page, key = read("STUDENT-CHECKS.md"), read("teacher/KEY-AND-NEXT.md")
    need(re.findall(r"^## Check ([AB]) · after Day (\d+) ·", page, re.MULTILINE) ==
         [("A", "5"), ("B", "10")], "two named fresh checks missing")
    need("## Check A worked response" in key and
         "## Check B worked response" in key,
         "separate worked check responses missing")
    for doc in (page, key):
        need("publicly accessible by url" in doc.lower() and
             "not secure exams" in doc.lower(),
             "public check/key security boundary absent")
    practice = "\n".join(read(name) for name in
                         ("SOURCE-CARDS.md", "LESSONS.md", "LEARNER.md"))
    for distinct in ("Mapfold Studio", "Cycle-repair club update",
                     "3 of 4 volunteer adult staff", "5 of 10 adult volunteers"):
        need(distinct.lower() not in practice.lower(),
             f"fresh check leaked into practice: {distinct}")
    for marker in ("3/4 = 75%", "5/10 = 50%", "possible access barrier",
                   "no implementation"):
        need(marker.lower() in key.lower() or
             (marker == "no implementation" and "no implementation" in page.lower()),
             f"worked check evidence/limit missing: {marker}")


def arithmetic(sources: dict[int, str]) -> None:
    d = Decimal
    need(d("8") / 12 * 100 > d("66.6") and d("8") / 12 * 100 < d("66.8") and
         d("4") / 12 * 100 > d("33.2") and d("4") / 12 * 100 < d("33.4") and
         d("3") / 4 * 100 == d("75") and d("5") / 10 * 100 == d("50"),
         "independent fictional fraction calculations drift")
    need(all(token in sources[9].lower() for token in
             ("12 volunteers", "8", "4", "no sampling frame")),
         "Day 9 source denominator/sampling boundary drift")
    page, key = read("STUDENT-CHECKS.md"), read("teacher/KEY-AND-NEXT.md")
    need(all(token in page for token in
             ("3 of 4 volunteer adult staff", "5 of 10 adult volunteers",
              "No young people were sampled", "not a representative population sample")),
         "fresh check givens/population limit drift")
    for token in ("8/12", "66.7%", "4/12", "33.3%", "3/4 = 75%", "5/10 = 50%"):
        need(token in key, f"published worked key missing {token}")
    rows = re.findall(r"^\| (\d\d) \| (.+) \| (.+) \|$", key, re.MULTILINE)
    need([int(day) for day, _, _ in rows] == DAYS,
         "ten daily source-bound key/next-move rows missing")


def assets() -> None:
    alternatives = read("print/TEXT-ALTERNATIVES.md")
    need(alternatives.count("**Tactile route:**") == 5 and all(
        title in alternatives for title in
        ("## Lens-and-language mat", "## Resource-and-access mat",
         "## Message-evidence mat", "## Dimension-map mat", "## Issue-question mat")),
        "five full text/tactile alternatives absent")
    ns = {"svg": "http://www.w3.org/2000/svg"}
    for stem in STEMS:
        svg = ROOT / "print" / f"{stem}.svg"
        pdf = svg.with_suffix(".pdf")
        image = ET.parse(svg).getroot()
        need(image.attrib.get("width") == "210mm" and
             image.attrib.get("height") == "297mm" and
             image.attrib.get("viewBox") == "0 0 210 297" and
             image.attrib.get("role") == "img" and
             image.attrib.get("aria-labelledby") == "title desc" and
             image.find("svg:title", ns) is not None and
             image.find("svg:desc", ns) is not None,
             f"{stem}: SVG A4/access labels absent")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        words = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        fonts = subprocess.check_output(["pdffonts", str(pdf)], text=True)
        need("(A4)" in info and re.search(r"Pages:\s+1\b", info) is not None and
             "SubjectNest" in words and "CC BY 4.0" in words and len(words) > 150 and
             "DejaVuSans" in fonts, f"{stem}: PDF page/text/font drift")
    with tempfile.TemporaryDirectory(prefix="subjectnest-y11-health-") as temp:
        subprocess.run([sys.executable, str(ROOT / "print/generate_print.py"),
                        "--output-dir", temp], capture_output=True, check=True, text=True)
        for stem in STEMS:
            for extension in ("svg", "pdf"):
                name = f"{stem}.{extension}"
                need(sha(Path(temp) / name) == sha(ROOT / "print" / name),
                     f"{name}: deterministic print regeneration drift")


def slug(heading: str) -> str:
    heading = re.sub(r"<[^>]+>", "", heading)
    return re.sub(r"[^a-z0-9_-]", "", heading.lower().strip().replace(" ", "-"))


def links_and_rights() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        content = path.read_text(encoding="utf-8")
        need("CC BY 4.0" in content, f"{path.name}: original-rights notice absent")
        for raw in re.findall(r"\[[^]]+\]\(([^)]+)\)", content):
            if raw.startswith(("https://", "http://", "mailto:")):
                continue
            relative, _, anchor = unquote(raw).partition("#")
            target = (path.parent / relative).resolve() if relative else path
            need(target.exists(), f"{path.name}: broken local link {raw}")
            if anchor and target.suffix.lower() == ".md":
                headings = [slug(m.group(1)) for m in re.finditer(
                    r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)]
                need(anchor in headings, f"{path.name}: broken fragment {raw}")
            count += 1
    return count


def files() -> list[Path]:
    return sorted((path for path in ROOT.rglob("*") if path.is_file() and
                   path != MANIFEST and "__pycache__" not in path.parts and
                   ".ruff_cache" not in path.parts),
                  key=lambda path: path.relative_to(ROOT).as_posix())


def manifest() -> str:
    return "".join(f"{sha(path)}  {path.relative_to(ROOT).as_posix()}\n" for path in files())


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    source_snapshot()
    sources = lesson_routes()
    checks()
    arithmetic(sources)
    assets()
    links = links_and_rights()
    expected = manifest()
    if args.write_manifest:
        MANIFEST.write_text(expected, encoding="utf-8")
    else:
        need(MANIFEST.read_text(encoding="utf-8") == expected, "SHA manifest drift")
    print(f"PASS: ten 25-minute scripts, 10 fictional cases, 30 switchable routes, "
          f"20 optional contexts, 2 fresh public checks/keys, 5 A4 SVG/PDF/text aids, "
          f"{links} local links, {len(files())} hashed files")


if __name__ == "__main__":
    main()
