#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Recheck Year 8 integrated W03–04 source rows, teaching, assets and receipt."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import tempfile
import warnings
import xml.etree.ElementTree as ET
from fractions import Fraction
from pathlib import Path
from urllib.parse import unquote

import openpyxl

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
FRAMEWORK = STUDIO / "data/frameworks/acara-v9.json"
WORKBOOK = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
MANIFEST = ROOT / "MANIFEST.sha256"
ROWS = {
    "AC9E8LA03": 937, "AC9E8LA04": 941, "AC9E8LE03": 967,
    "AC9E8LY03": 994, "AC9E8LY06": 1007, "AC9E8LY07": 1012,
    "AC9HE8K01": 1995, "AC9HE8S03": 2029, "AC9HE8S04": 2034,
    "AC9M8A01": 18131, "AC9M8A02": 18135, "AC9M8A03": 18142,
    "AC9S8U05": 19305, "AC9S8I01": 19357, "AC9S8I02": 19364,
    "AC9S8I03": 19370, "AC9S8I04": 19377, "AC9S8I06": 19392,
    "AC9TDE8K06": 19966, "AC9TDE8P01": 19976, "AC9TDE8P02": 19985,
    "AC9TDE8P04": 19999, "AC9TDI8P07": 20363, "AC9TDI8P10": 20379,
    "AC9AVA8D01": 21631, "AC9AVA8C01": 21648, "AC9AVA8C02": 21655,
}
BOUNDARIES = {
    "AC9E8LA03": "One fictional hybrid notice; not the range of text structures.",
    "AC9E8LA04": "One compact source paragraph; not broad paragraph cohesion.",
    "AC9E8LE03": "One invented artist note and reader test; no population effect claim.",
    "AC9E8LY03": "Selected invented source and quotation checks only.",
    "AC9E8LY06": "Draft, edit and private audience test; publication breadth remains.",
    "AC9E8LY07": "Only an actually delivered spoken route may evidence voice; AAC/writing logged separately.",
    "AC9M8A01": "Repeated linear paper expressions; not every property/form.",
    "AC9M8A02": "Whole-number inequalities and table/substitution only; graph breadth remains.",
    "AC9M8A03": "Fictional width, time and quote models with stated domains; limited model review.",
    "AC9S8U05": "Only a light paper tab actually observed moving; no quantified energy study.",
    "AC9S8I01": "One fold-shape question and reasoned prediction.",
    "AC9S8I02": "Conditional on actually running/direction of local safe trial; paper plan alone is not conduct.",
    "AC9S8I03": "Only an actual safe ruler/tab trial evidences equipment use; fallback analysis does not.",
    "AC9S8I04": "One small own or marked constructed two-row trial table.",
    "AC9S8I06": "Limited critique of crease, placement and claim transfer.",
    "AC9HE8K01": "Invented competing quotes and one price change; not real market evidence.",
    "AC9HE8S03": "Two fictional supplier cards; no broad economic data trend.",
    "AC9HE8S04": "One model decision with possible costs/benefits; no real action.",
    "AC9TDE8K06": "Paper fold/form discussion only; no real mounting/material certification.",
    "AC9TDE8P01": "Small paper design brief and material selection; real requirements untested.",
    "AC9TDE8P02": "Only learner-created/tested/revised paper idea evidences iteration.",
    "AC9TDE8P04": "Limited paper criteria, with sustainability a question rather than life-cycle proof.",
    "AC9TDI8P07": "Reader-story/interface design; paper wireframe route is design reasoning.",
    "AC9TDI8P10": "Only actual tool/user-story evaluation counts; paper simulation is not browser test.",
    "AC9AVA8D01": "Two original paper spacing studies, not broad media/material techniques.",
    "AC9AVA8C01": "Two small original documented caption concepts.",
    "AC9AVA8C02": "Only an actually made/revised original caption composition counts.",
}
ASSETS = ("two-rail-claim", "paper-test-card", "decision-board")
TEACHING = ("README.md", "SOURCES.md", "LESSONS.md", "LEARNER.md", "EXTRAS-AND-HOME.md",
            "interactive/TEXT-ROUTE.md", "print/TEXT-ALTERNATIVES.md")


def need(condition: bool, message: str) -> None:
    if not condition:
        raise AssertionError(message)


def sha(path: Path) -> str:
    return hashlib.sha256(path.read_bytes()).hexdigest()


def official() -> dict[str, dict]:
    need(sha(WORKBOOK) == SOURCE_SHA, "official workbook SHA mismatch")
    data = json.loads(FRAMEWORK.read_text(encoding="utf-8"))
    need(data["framework"] == "Australian Curriculum Version 9.0" and
         data["source_sha256"] == SOURCE_SHA and data["retrieved_at"] == "2026-09-29",
         "framework provenance/version mismatch")
    selected = [r for r in data["records"] if r.get("record_type") == "content_description" and
                r.get("code") in ROWS]
    need(len(selected) == len(ROWS) == 27, "27 unique official descriptions required")
    by_code = {r["code"]: r for r in selected}
    with warnings.catch_warnings():
        warnings.simplefilter("ignore", UserWarning)
        sheet = openpyxl.load_workbook(WORKBOOK, read_only=True, data_only=True)["Learning areas"]
        source_rows = {i: tuple(row) for i, row in enumerate(sheet.iter_rows(values_only=True), 1)
                       if i in set(ROWS.values())}
    for code, row_number in ROWS.items():
        record = by_code[code]
        attr = record["attributes"]
        original = source_rows[row_number]
        need(record["source_row"] == row_number and original[4] == code, f"{code}: workbook row/code drift")
        need(attr["code"] == code and attr["learning_area"] == original[0] and
             attr["level"] == original[2] and attr["subject"] == original[1],
             f"{code}: area/level/subject drift")
        need(" ".join(record["plain_text"].split()) ==
             " ".join(attr["content_description"].split()) ==
             " ".join(original[9].split()), f"{code}: exact wording drift")
        need(attr["level"] in ("Year 8", "Years 7 and 8"), f"{code}: wrong band")
    return by_code


def sections() -> dict[int, tuple[str, str]]:
    source = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) Day (\d+) (.+)$", source, re.MULTILINE))
    need([int(m.group(2)) for m in marks] == list(range(11, 21)), "ten sequential lesson headers")
    result = {}
    used = set()
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (3 if day <= 15 else 4), f"Day {day}: week mismatch")
        body = source[mark.end():marks[index + 1].start() if index + 1 < len(marks) else len(source)]
        minutes = [int(m) for m in re.findall(r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*", body, re.MULTILINE)]
        need(minutes == [3, 5, 7, 12, 5, 3], f"Day {day}: 35-minute six-step rhythm {minutes}")
        need("**Goal:**" in body and "**Prepare:**" in body and
             ("**Exit" in body or (day in (15, 20) and "**Collect" in body)),
             f"Day {day}: goal/preparation/exit missing")
        codes = set(re.findall(r"\bAC9[A-Z0-9]+\b", body))
        need(codes and codes <= ROWS.keys(), f"Day {day}: no/invalid codes {codes - ROWS.keys()}")
        used |= codes
        result[day] = (mark.group(3), body)
    need(used == ROWS.keys(), f"lesson/crosswalk code mismatch {used ^ ROWS.keys()}")
    need("not independent print decoding" in source and "constructed data" in source.lower(),
         "access or constructed-data boundary missing")
    return result


def crosswalk(by_code: dict[str, dict], lessons: dict[int, tuple[str, str]]) -> str:
    lines = ["# ACARA v9 crosswalk Year 8 integrated Weeks 3–4", "",
             "**Pinned source:** [ACARA Version 9.0 official workbook](https://www.australiancurriculum.edu.au/downloads), accessed 29 September 2026; original XLSX SHA-256 `" + SOURCE_SHA + "`. Each row below is checked against the actual workbook cell and local import `products/curriculum-studio/data/frameworks/acara-v9.json`. Exact levels preserve **Year 8** or **Years 7 and 8**. These are partial or conditional lesson opportunities, not complete descriptions, state/territory approval or achieved outcomes. See [source and rights ledger](SOURCES-AND-RIGHTS.md).", "",
             "| Code | Workbook row | Exact level | Area / subject | Exact official content description | Planned day/action | Boundary |",
             "| --- | ---: | --- | --- | --- | --- | --- |"]
    for code, row in sorted(ROWS.items(), key=lambda pair: pair[1]):
        record = by_code[code]
        attr = record["attributes"]
        desc = " ".join(record["plain_text"].split()).replace("|", "\\|")
        days = ", ".join(f"{day} ({title})" for day, (title, body) in lessons.items() if code in body)
        need(days, f"{code}: no lesson evidence")
        lines.append(f"| `{code}` | {row} | {attr['level']} | {attr['learning_area']} / {attr['subject']} | {desc} | {days} | {BOUNDARIES[code]} |")
    lines += ["", "## Reading the crosswalk", "",
              "A code means an **opportunity for part of its description**. Source C's constructed paper data are not observed class results. Conduct/equipment/energy codes depend on what the learner actually performs or observes. Digital Technologies browser-testing evidence requires actual interaction with the offline tool; a paper wireframe gives design reasoning only. Visual Arts creation needs an actual original composition. Spoken English evidence requires an actual spoken route, with AAC and written routes recorded separately. No financial, safety, attendance, language/cultural or classroom outcome claim follows from these paper models.", "",
              "Broader literature, Year 8 science knowledge, authentic HASS inquiry, digital implementation, local cultural authority and state/territory syllabuses require separate teaching and review. The two fresh checks provide narrow transfer samples, not achievement-standard decisions.", "",
              "**ACARA attribution:** © Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. Selected wording was downloaded from the [Australian Curriculum website](https://www.australiancurriculum.edu.au/downloads) (accessed 29 September 2026), selected and plain-text whitespace normalised, under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/) subject to [ACARA terms](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse SubjectNest. Original commentary © NeuroForgeIO Pty Ltd 2026, CC BY 4.0.", ""]
    return "\n".join(lines)


def learners_checks_arithmetic() -> None:
    learner = (ROOT / "LEARNER.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)$", learner, re.MULTILINE))
    need([int(m.group(1)) for m in marks] == list(range(11, 21)), "ten learner days")
    for index, mark in enumerate(marks):
        day = int(mark.group(1))
        body = learner[mark.end():marks[index + 1].start() if index + 1 < len(marks) else len(learner)]
        need(re.findall(r"^- \*\*([ABC]) ", body, re.MULTILINE) == ["A", "B", "C"],
             f"Day {day}: exactly three A/B/C routes")
        need("**Exit:**" in body, f"Day {day}: missing exit")
    need("ASSESSMENT.md" not in learner and "TEACHER-KEY.md" not in learner,
         "learner page exposes held-out check or key")
    extras = (ROOT / "EXTRAS-AND-HOME.md").read_text(encoding="utf-8")
    rows = re.findall(r"^\| (\d{2}) \| (.+) \| (.+) \| (.+) \|$", extras, re.MULTILINE)
    need([int(r[0]) for r in rows] == list(range(11, 21)) and
         all(all(cell.strip() for cell in row[1:]) for row in rows), "20 extras + 10 home routes")
    need("no device, account, purchase" in extras, "home no-purchase boundary")
    teaching = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in TEACHING)
    for phrase in ("3(8+5k)", "every real noticeboard", "38-minute model session", "H(c)=10+3c",
                   "everyone will finish a zine"):
        need(phrase.lower() not in teaching.lower(), f"held-out phrase leaked into teaching: {phrase}")
    checks = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    need(all(x in checks for x in ("Check A Day 15", "Check B Day 20", "3(8+5k)", "H(c)=10+3c")),
         "two fresh assessment sources missing")
    for code in ("A1", "A2", "A3", "A4", "B1", "B2", "B3", "B4"):
        need(re.search(rf"^\| {code} ·", key, re.MULTILINE) is not None, f"rubric {code} missing")
    need("not independent print decoding" in checks and "not a real mounting" in teaching,
         "assessment/access/site boundary missing")
    need(2 * (12 + 18 * 4 + 12) == 192 and 24 + 18 * 5 == 114 and
         2 * (12 + 18 * 5 + 12) == 228 and
         Fraction(1 + 2 + 1, 3) == Fraction(4, 3) and
         Fraction(4 + 4 + 5, 3) == Fraction(13, 3) and
         5 + 15 * 3 == 50 and 5 + 15 * 4 == 65 and
         12 + 2 * 4 == 20 and 4 * 4 == 16 and
         12 + 2 * 8 == 28 and 4 * 8 == 32 and
         12 + 2 * 6 == 4 * 6 == 24 and 6 * 4 == 24 and
         3 * (8 + 5 * 5) == 24 + 15 * 5 == 99 and 8 + 5 * 6 == 38 and
         6 + 8 * 4 == 38 and 6 + 8 * 5 == 46 and
         10 + 3 * 4 == 22 and 5 * 4 == 20,
         "source/check arithmetic invariant failed")


def assets_and_tool() -> None:
    alt = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    need(all(h in alt for h in ("## Two rail claim mat", "## Paper test card", "## Decision board")),
         "three full text/tactile alternatives")
    ns = {"s": "http://www.w3.org/2000/svg"}
    for stem in ASSETS:
        svg = ROOT / "print" / f"{stem}.svg"
        pdf = svg.with_suffix(".pdf")
        root = ET.parse(svg).getroot()
        need(root.attrib.get("width") == "210mm" and root.attrib.get("height") == "297mm" and
             root.attrib.get("viewBox") == "0 0 210 297" and root.attrib.get("role") == "img" and
             root.attrib.get("aria-labelledby") == "title desc", f"{stem}: SVG A4/access data")
        need(root.find("s:title", ns) is not None and root.find("s:desc", ns) is not None,
             f"{stem}: title/desc")
        need("CC BY 4.0" in svg.read_text(encoding="utf-8"), f"{stem}: SVG rights")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        extracted = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        fonts = subprocess.check_output(["pdffonts", str(pdf)], text=True)
        need("(A4)" in info and re.search(r"Pages:\s+1\b", info) is not None and
             "SubjectNest" in extracted and "CC BY 4.0" in extracted and len(extracted) > 180 and
             "DejaVuSans" in fonts, f"{stem}: PDF page/text/font")
    with tempfile.TemporaryDirectory(prefix="subjectnest-y8-integrated-") as tmp:
        subprocess.run([sys.executable, str(ROOT / "print/generate_print.py"), "--output-dir", tmp],
                       check=True, capture_output=True, text=True)
        for stem in ASSETS:
            for suffix in (".svg", ".pdf"):
                name = stem + suffix
                need(sha(Path(tmp) / name) == sha(ROOT / "print" / name),
                     f"{name}: nondeterministic render")
    tool = (ROOT / "interactive/model-audit.html").read_text(encoding="utf-8")
    need(all(x in tool for x in ("id=\"audit-form\"", "id=\"result\"", "aria-live=\"polite\"",
                                  "Source B", "Source D", "Source E", "Paper model only.")),
         "offline tool fields/source/limit missing")
    need(not any(x in tool for x in ("fetch(", "XMLHttpRequest", "localStorage", "sessionStorage",
                                     "sendBeacon", "<script src=", "<link rel=\"stylesheet\"")),
         "tool must be offline and storage-free")


def slug(heading: str) -> str:
    heading = re.sub(r"<[^>]+>", "", heading)
    return re.sub(r"[^a-z0-9-]", "", heading.strip().lower().replace(" ", "-"))


def links_and_rights() -> int:
    checked = 0
    for path in ROOT.rglob("*.md"):
        source = path.read_text(encoding="utf-8")
        need("CC BY 4.0" in source, f"{path}: licence/attribution absent")
        for raw in re.findall(r"\[[^]]+\]\(([^)]+)\)", source):
            if raw.startswith(("https://", "http://", "mailto:")):
                continue
            relative, _, anchor = unquote(raw).partition("#")
            target = (path.parent / relative).resolve() if relative else path
            need(target.exists(), f"{path}: broken link {raw}")
            if anchor and target.suffix.lower() == ".md":
                headings = [slug(m.group(1)) for m in re.finditer(
                    r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)]
                need(anchor in headings, f"{path}: broken heading #{anchor} in {target}")
            checked += 1
    html = (ROOT / "interactive/model-audit.html").read_text(encoding="utf-8")
    for raw in re.findall(r'href="([^"]+)"', html):
        if raw.startswith(("https://", "http://", "#")):
            continue
        need((ROOT / "interactive" / raw).exists(), f"HTML broken local href {raw}")
        checked += 1
    return checked


def payload_files() -> list[Path]:
    return sorted((p for p in ROOT.rglob("*") if p.is_file() and p != MANIFEST and
                   "__pycache__" not in p.parts and ".ruff_cache" not in p.parts),
                  key=lambda p: p.relative_to(ROOT).as_posix())


def manifest() -> str:
    return "".join(f"{sha(p)}  {p.relative_to(ROOT).as_posix()}\n" for p in payload_files())


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    by_code = official()
    lessons = sections()
    expected_crosswalk = crosswalk(by_code, lessons)
    crosswalk_path = ROOT / "CURRICULUM-CROSSWALK.md"
    if args.write_crosswalk:
        crosswalk_path.write_text(expected_crosswalk, encoding="utf-8")
    else:
        need(crosswalk_path.read_text(encoding="utf-8") == expected_crosswalk,
             "exact official crosswalk/day drift")
    learners_checks_arithmetic()
    assets_and_tool()
    link_count = links_and_rights()
    expected_manifest = manifest()
    if args.write_manifest:
        MANIFEST.write_text(expected_manifest, encoding="utf-8")
    else:
        need(MANIFEST.read_text(encoding="utf-8") == expected_manifest, "SHA manifest drift")
    print(f"PASS: 10 × 35-minute lessons; 30 routes; 20 extras + 10 home routes; "
          f"2 fresh checks/8 criteria; 27 exact ACARA workbook rows; "
          f"3 A4 SVG/PDF/text aids; offline tool; {link_count} links; {len(payload_files())} hashed files")


if __name__ == "__main__":
    main()
