#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Fail closed on local Year 11 integrated source, lesson, arithmetic and asset drift."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import tempfile
import xml.etree.ElementTree as ET
from decimal import Decimal
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
MANIFEST = ROOT / "MANIFEST.sha256"
DAYS = list(range(11, 21))
STEMS = ("source-claim", "rate-base", "budget-window", "percent-period")
QCAA = "https://www.qcaa.qld.edu.au"


def need(condition: bool, message: str) -> None:
    if not condition:
        raise SystemExit(f"FAIL: {message}")


def sha(path: Path) -> str:
    return hashlib.sha256(path.read_bytes()).hexdigest()


def sections(path: str) -> dict[int, str]:
    data = (ROOT / path).read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+) · (.+)$", data, re.MULTILINE))
    need([int(m.group(1)) for m in marks] == DAYS, f"{path}: expected Days 11–20 in order")
    return {int(mark.group(1)): data[mark.end():marks[i + 1].start()
                                      if i + 1 < len(marks) else len(data)]
            for i, mark in enumerate(marks)}


def source_snapshot() -> None:
    s = json.loads((ROOT / "source-snapshot.json").read_text(encoding="utf-8"))
    need(s["checked_on"] == "2026-09-29" and s["jurisdiction"] == "Queensland, Australia",
         "source snapshot date/jurisdiction drift")
    en, gm, qce, ato = (s[k] for k in ("english", "general_mathematics", "qce_handbook", "ato_gst_rule"))
    need(en["title"] == "English 2025 v1.3" and
         en["pdf_url"] == QCAA + "/downloads/senior-qce/syllabuses/snr_english_25_syll.pdf" and
         en["landing_url"] == QCAA + "/senior/senior-subjects/syllabuses/english/english" and
         en["unit"] == "Unit 1: Perspectives and texts" and
         en["printed_pages"] == {"unit_overview": 17, "objectives": 18,
                                 "subject_matter_start": 19, "subject_matter_end": 20,
                                 "unit_1_2_reporting": 16},
         "QCAA English version/location drift")
    need(gm["title"] == "General Mathematics 2025 v1.3" and
         gm["pdf_url"] == QCAA + "/downloads/senior-qce/syllabuses/snr_maths_general_25_syll.pdf" and
         gm["landing_url"] == QCAA + "/senior/senior-subjects/syllabuses/mathematics/general-mathematics" and
         gm["unit"] == "Unit 1: Money, measurement, algebra and linear equations" and
         gm["topic"] == "Topic 1: Consumer arithmetic" and
         gm["topic_notional_hours"] == 14 and gm["printed_subject_matter_page"] == 15,
         "QCAA Mathematics version/location drift")
    need(en["pdf_publication"] == gm["pdf_publication"] == "January 2026" and
         en["implementation"] == gm["implementation"] == "students completing in 2026 or beyond",
         "publication/cohort drift")
    need(qce["general_unit_notional_hours"] == 55 and
         qce["url"] == QCAA + "/senior/certificates-and-qualifications/qce-qcia-handbook/4-qld-curriculum/4.1-syllabuses" and
         ato["url"].startswith("https://www.ato.gov.au/law/view/document?locid="),
         "QCE/ATO authority drift")
    crosswalk = (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8")
    ledger = (ROOT / "SOURCE-AND-RIGHTS.md").read_text(encoding="utf-8")
    for marker in ("2025 v1.3", "29 September 2026", "Topic 1", "14 notional hours",
                   "55 hours", "school decisions", "not secure exams"):
        need(marker.lower() in crosswalk.lower(), f"crosswalk missing scope: {marker}")
    for entry in (en, gm):
        need(entry["pdf_url"] in crosswalk and entry["pdf_url"] in ledger,
             f"official PDF absent from crosswalk/ledger: {entry['title']}")
    need(ato["url"] in ledger and qce["url"] in crosswalk,
         "ATO/QCE references missing")
    need(re.search(r"\bAC9[A-Z0-9]{5,}\b", crosswalk) is None,
         "invented senior AC9 code")
    need("not future web freshness" in s["source_archive_policy"] and
         "not redistributed" in ledger.lower(),
         "source-byte/live-freshness limitation missing")


def lesson_routes() -> tuple[dict[int, str], dict[int, str]]:
    sources = sections("SOURCE-CARDS.md")
    lessons = sections("LESSONS.md")
    learners = sections("LEARNER.md")
    for day in DAYS:
        need("**Public" in sources[day] and "**Desk note" in sources[day] and
             "**Inquiry:**" in sources[day], f"Day {day}: complete source pair absent")
        need(all(marker in lessons[day] for marker in ("**0–4", "**4–9", "**9–16",
                                                        "**16–27", "**27–32", "**32–35")),
             f"Day {day}: six 35-minute phases absent")
        routes = re.findall(r"^- \*\*([ABC]) · ([^*]+):\*\* (.+)$", learners[day], re.MULTILINE)
        need([r[0] for r in routes] == ["A", "B", "C"] and
             all(len(r[2]) >= 90 for r in routes) and
             "**Optional extra / no-purchase home:**" in learners[day],
             f"Day {day}: three substantial routes/home missing")
    intro = (ROOT / "LEARNER.md").read_text(encoding="utf-8").split("## Day 11")[0]
    need(all(marker in intro.lower() for marker in
             ("same three things", "not fixed learning-style", "read-aloud")) or
         all(marker in intro.lower() for marker in
             ("same three things", "not fixed learning-style labels", "read-aloud")),
         "shared target/access boundary missing")
    return sources, lessons


def checks() -> None:
    page = (ROOT / "STUDENT-CHECKS.md").read_text(encoding="utf-8")
    key = (ROOT / "teacher/KEY-AND-NEXT.md").read_text(encoding="utf-8")
    need(re.findall(r"^## Check ([AB]) · Day (\d+) ·", page, re.MULTILINE) ==
         [("A", "15"), ("B", "20")], "two Day 15/20 fresh checks missing")
    need("## Check A worked response" in key and "## Check B worked response" in key,
         "worked check key missing")
    for doc in (page, key):
        need("publicly accessible" in doc.lower() and "not secure exam" in doc.lower(),
             "published check/key security limit absent")
    practice = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in
                         ("SOURCE-CARDS.md", "LESSONS.md", "LEARNER.md"))
    for distinct in ("Origami display sheets", "Ceramic-pattern paper cards",
                     "Fold Gallery coordinator", "Paper Kiln editor"):
        need(distinct.lower() not in practice.lower(), f"fresh check leaked to practice: {distinct}")
    for token in ("$1.30/sheet", "$1.15/sheet", "$15.60", "$27.60", "$12.00",
                  "$31.20", "$3.60", "$97", "$92", "$5"):
        need(token in key, f"check arithmetic/key token missing: {token}")


def arithmetic(sources: dict[int, str]) -> None:
    d = Decimal
    actual = {
        11: (d("20") / d("25"), d("35") / d("50"), d("20"), d("40"), d("35")),
        12: (d("58240") / d("52"),),
        13: (d("12") + d("0.40") * 40, d("0.65") * 40,
             d("12") + d("0.40") * 60, d("0.65") * 60),
        14: (d("140") + d("90"), d("380") - d("140") - d("90"),
             d("380") - d("140") - d("90") - d("80")),
        15: (d("1200") * d("0.03") * 2, d("1200") * (1 + d("0.03") * 2)),
        16: (d("120") * d("0.15"), d("120") * d("0.85"), d("99") + d("8")),
        17: (d("24") * 4, d("11") * 8),
        18: (d("75") * d("0.10"), d("75") * d("1.10"), d("86") - d("82.50")),
        19: (d("210") - d("120") - d("55"),
             4 * (d("210") - d("120") - d("55"))),
        20: (d("40") * d("1.25"), d("40") * d("1.25") * d("0.90")),
    }
    expect = {
        11: ("0.80", "0.70", "20", "40", "35"),
        12: ("1120",),
        13: ("28", "26", "36", "39"),
        14: ("230", "150", "70"),
        15: ("72", "1272"),
        16: ("18", "102", "107"),
        17: ("96", "88"),
        18: ("7.50", "82.50", "3.50"),
        19: ("35", "140"),
        20: ("50", "45"),
    }
    source_tokens = {
        11: ("25 cards for $20", "50 cards for $35", "40 cards", "20"),
        12: ("$58,240", "52 weeks"),
        13: ("$12 fixed", "$0.40 per card", "$0.65 per card", "40", "60"),
        14: ("$380", "$140", "$90", "$80"),
        15: ("$1,200", "3% per year", "2 years"),
        16: ("$120", "15%", "$99", "$8"),
        17: ("$24 per hour", "4 hours", "$11 per accepted sleeve", "8 accepted sleeves"),
        18: ("$75 GST-exclusive", "$86 GST-inclusive", "fully taxable"),
        19: ("$210", "$120 fixed", "$55 discretionary", "four-week"),
        20: ("$40", "25%", "10%"),
    }
    key = (ROOT / "teacher/KEY-AND-NEXT.md").read_text(encoding="utf-8")
    rows = {int(day): result for day, result, _ in re.findall(
        r"^\| (\d\d) \| (.+) \| (.+) \|$", key, re.MULTILINE)}
    need(sorted(map(int, rows)) == DAYS, "daily worked-key rows missing")
    for day in DAYS:
        need(actual[day] == tuple(d(v) for v in expect[day]),
             f"Day {day}: arithmetic drift")
        need(all(token in sources[day] for token in source_tokens[day]),
             f"Day {day}: source-model fact drift")
        # Require every computed number in the published worked key, with a dollar marker or a unit.
        for number in set(expect[day]):
            need(re.search(rf"(?<!\d){re.escape(number)}(?!\d)",
                           rows[day].replace(",", "")) is not None,
                 f"Day {day}: answer missing {number}")
    need(d("15.60") / 12 == d("1.30") and d("27.60") / 24 == d("1.15") and
         2 * d("15.60") == d("31.20") and d("31.20") - d("27.60") == d("3.60") and
         d("90") + d("7") == d("97") and d("100") * d("0.92") == d("92"),
         "check A/B arithmetic drift")


def assets() -> None:
    alternatives = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    need(alternatives.count("**Tactile route:**") == 4 and
         all(title in alternatives for title in ("## Source-claim mat", "## Rate-base mat",
                                                  "## Budget-window mat", "## Percent-period mat")),
         "four full text/tactile A4 routes missing")
    ns = {"s": "http://www.w3.org/2000/svg"}
    for stem in STEMS:
        svg = ROOT / "print" / f"{stem}.svg"
        pdf = svg.with_suffix(".pdf")
        art = ET.parse(svg).getroot()
        need(art.attrib.get("width") == "210mm" and art.attrib.get("height") == "297mm" and
             art.attrib.get("viewBox") == "0 0 210 297" and art.attrib.get("role") == "img" and
             art.attrib.get("aria-labelledby") == "title desc" and
             art.find("s:title", ns) is not None and art.find("s:desc", ns) is not None,
             f"{stem}: SVG geometry/access labels")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        words = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        fonts = subprocess.check_output(["pdffonts", str(pdf)], text=True)
        need("(A4)" in info and re.search(r"Pages:\s+1\b", info) is not None and
             "SubjectNest" in words and "CC BY 4.0" in words and len(words) > 190 and
             "DejaVuSans" in fonts, f"{stem}: PDF size/text/font")
    with tempfile.TemporaryDirectory(prefix="subjectnest-y11-integrated-") as tmp:
        subprocess.run([sys.executable, str(ROOT / "print/generate_print.py"), "--output-dir", tmp],
                       check=True, capture_output=True, text=True)
        for stem in STEMS:
            for suffix in (".svg", ".pdf"):
                name = stem + suffix
                need(sha(Path(tmp) / name) == sha(ROOT / "print" / name),
                     f"{name}: print reproducibility drift")
    html = (ROOT / "interactive/whole-set-lab.html").read_text(encoding="utf-8")
    need(all(marker in html for marker in ("id=\"set-form\"", "id=\"result\"",
                                         "aria-live=\"polite\"", "max=\"1000\"",
                                         "const cost = 20 * small + 35 * large")),
         "offline tool input/result/math markers absent")
    need(not any(marker in html for marker in ("fetch(", "XMLHttpRequest", "localStorage",
                                                "sessionStorage", "sendBeacon", "<script src=",
                                                "<link rel=\"stylesheet\"")),
         "offline tool network/storage dependency")
    paper = (ROOT / "interactive/TEXT-ROUTE.md").read_text(encoding="utf-8")
    need(all(marker in paper for marker in ("20 | 1 S", "40 | 2 S", "75 | 3 S",
                                               "1 S + 1 L", "**Tactile route:**")),
         "interactive full paper route incomplete")


def slug(heading: str) -> str:
    heading = re.sub(r"<[^>]+>", "", heading)
    return re.sub(r"[^a-z0-9_-]", "", heading.lower().strip().replace(" ", "-"))


def links_and_rights() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        source = path.read_text(encoding="utf-8")
        need("CC BY 4.0" in source, f"{path.name}: original rights notice missing")
        for raw in re.findall(r"\[[^]]+\]\(([^)]+)\)", source):
            if raw.startswith(("https://", "http://", "mailto:")):
                continue
            rel, _, anchor = unquote(raw).partition("#")
            target = (path.parent / rel).resolve() if rel else path
            need(target.exists(), f"{path.name}: broken link {raw}")
            if anchor and target.suffix.lower() == ".md":
                headings = [slug(m.group(1)) for m in re.finditer(
                    r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)]
                need(anchor in headings, f"{path.name}: broken fragment {raw}")
            count += 1
    html = (ROOT / "interactive/whole-set-lab.html").read_text(encoding="utf-8")
    for raw in re.findall(r'href="([^"]+)"', html):
        if raw.startswith(("https://", "http://", "#")):
            continue
        rel, _, anchor = raw.partition("#")
        target = (ROOT / "interactive" / rel).resolve()
        need(target.exists(), f"HTML broken local link {raw}")
        if anchor and target.suffix.lower() == ".md":
            headings = [slug(m.group(1)) for m in re.finditer(
                r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)]
            need(anchor in headings, f"HTML broken fragment {raw}")
        count += 1
    return count


def files() -> list[Path]:
    return sorted((path for path in ROOT.rglob("*") if path.is_file() and path != MANIFEST and
                   "__pycache__" not in path.parts and ".ruff_cache" not in path.parts),
                  key=lambda path: path.relative_to(ROOT).as_posix())


def manifest() -> str:
    return "".join(f"{sha(path)}  {path.relative_to(ROOT).as_posix()}\n" for path in files())


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    source_snapshot()
    sources, _ = lesson_routes()
    checks()
    arithmetic(sources)
    assets()
    links = links_and_rights()
    expected = manifest()
    if args.write_manifest:
        MANIFEST.write_text(expected, encoding="utf-8")
    else:
        need(MANIFEST.read_text(encoding="utf-8") == expected, "SHA manifest drift")
    print(f"PASS: 10 × 35-minute scripts, 30 switchable routes, 10 optional extras, "
          f"2 public fresh checks/keys, 4 A4 SVG/PDF/text aids, offline lab, "
          f"{links} local links, {len(files())} hashed files")


if __name__ == "__main__":
    main()
