#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Check the optional Year 4 menus, pinned codes, files and SHA-256 receipt."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
from pathlib import Path

from build_content import CARDS, render

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
FRAMEWORK = STUDIO / "data/frameworks/acara-v9.json"
WORKBOOK = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
MANIFEST = ROOT / "MANIFEST.sha256"
SUBJECTS = ("science", "hass", "hpe", "technologies", "arts")
ROWS = {
    "science": {"AC9S4U03": 18865, "AC9S4U04": 18875, "AC9S4I01": 18900, "AC9S4I02": 18906,
                "AC9S4I03": 18911, "AC9S4I04": 18917, "AC9S4I05": 18923, "AC9S4I06": 18931},
    "hass": {"AC9HS4S01": 1498, "AC9HS4S02": 1503, "AC9HS4S03": 1509, "AC9HS4S04": 1514,
             "AC9HS4S05": 1520, "AC9HS4S06": 1524, "AC9HS4S07": 1530},
    "hpe": {"AC9HP4P03": 3311, "AC9HP4P04": 3317, "AC9HP4P05": 3323, "AC9HP4P07": 3336,
            "AC9HP4M01": 3364, "AC9HP4M02": 3370, "AC9HP4M03": 3375, "AC9HP4M07": 3395,
            "AC9HP4M08": 3399, "AC9HP4M09": 3404},
    "technologies": {"AC9TDE4K02": 19803, "AC9TDE4P01": 19823, "AC9TDE4P02": 19830,
                     "AC9TDE4P03": 19836, "AC9TDE4P04": 19843, "AC9TDE4P05": 19849,
                     "AC9TDI4K03": 20175, "AC9TDI4P01": 20182, "AC9TDI4P02": 20189,
                     "AC9TDI4P04": 20200, "AC9TDI4P05": 20208},
    "arts": {"AC9ADA4D01": 20597, "AC9ADA4C01": 20604, "AC9ADA4P01": 20611,
             "AC9ADR4D01": 20821, "AC9ADR4C01": 20827, "AC9ADR4P01": 20836,
             "AC9AMA4D01": 21053, "AC9AMA4C01": 21060, "AC9AMA4P01": 21068,
             "AC9AMU4D01": 21309, "AC9AMU4C01": 21318, "AC9AMU4P01": 21328,
             "AC9AVA4D01": 21556, "AC9AVA4C01": 21563, "AC9AVA4P01": 21571},
}
AREA = {"science": "Science", "hass": "Humanities and Social Sciences", "hpe": "Health and Physical Education",
        "technologies": "Technologies", "arts": "The Arts"}
LEVEL = {"science": "Year 4", "hass": "Year 4", "hpe": "Years 3 and 4",
         "technologies": "Years 3 and 4", "arts": "Years 3 and 4"}
LIMIT = {
    "science": "Model values are labelled invented. Actual investigation, instrument use and force/material claims require approved observation and controls.",
    "hass": "H1–H5 are invented inquiry sources; no real place, event, local perspective or community response is established.",
    "hpe": "Paper or fictional role work is separate from actual voluntary movement, consent and group participation.",
    "technologies": "A design plan is separate from a tested product; an unplugged algorithm is separate from a run visual program.",
    "arts": "Composing/planning is separate from actual movement, enactment, media making, sounding and voluntary sharing.",
}


def check(condition: bool, message: str) -> None:
    if not condition:
        raise AssertionError(message)


def digest(path: Path) -> str:
    return hashlib.sha256(path.read_bytes()).hexdigest()


def framework_rows() -> dict[str, dict]:
    data = json.loads(FRAMEWORK.read_text(encoding="utf-8"))
    check(data["source_sha256"] == SOURCE_SHA, "framework workbook receipt drift")
    check(digest(WORKBOOK) == SOURCE_SHA, "original ACARA workbook byte hash drift")
    codes = {code for group in ROWS.values() for code in group}
    records = [r for r in data["records"] if r.get("code") in codes]
    check(len(records) == len(codes) == 51, "expected 51 distinct official descriptions")
    by_code = {r["code"]: r for r in records}
    for subject, expected in ROWS.items():
        used = {code for card in CARDS[subject] for code in card.codes.split()}
        check(used == set(expected), f"{subject}: selected code inventory drift")
        for code, source_row in expected.items():
            row = by_code[code]
            attr = row["attributes"]
            check(row["source_row"] == source_row, f"{code}: workbook row drift")
            check(attr["level"] == LEVEL[subject], f"{code}: level drift")
            check(attr["learning_area"] == AREA[subject], f"{code}: area drift")
            check(" ".join(row["plain_text"].split()) == " ".join(attr["content_description"].split()),
                  f"{code}: official wording mismatch")
            if subject == "technologies":
                check(attr["subject"] == ("Design and Technologies" if code.startswith("AC9TDE") else "Digital Technologies"),
                      f"{code}: technology strand drift")
    return by_code


def crosswalk(by_code: dict[str, dict]) -> str:
    lines = [
        "# ACARA v9 crosswalk · Year 4 optional Weeks 3–4 menus", "",
        "**Pinned national source:** [ACARA Australian Curriculum v9 downloads](https://www.australiancurriculum.edu.au/downloads), original workbook accessed 29 September 2026; SHA-256 `" + SOURCE_SHA + "`. Each source row below is the workbook row stored in the [local framework](../../../../../data/frameworks/acara-v9.json). Wording is exact after plain-text whitespace normalisation; no elaboration text is substituted. Science and HASS are **Year 4**; HPE, Technologies and The Arts are official **Years 3 and 4** bands. These are selected short opportunities, not a complete allocation, evidence of achievement, or jurisdictional approval.", "",
        "**Evidence rule:** Each day links an opportunity to a content description. A learner shows only the action observed. MODEL/PLAN/paper routes must not be promoted to a REAL investigation, bodily performance, actual media construction, sounded music or executed visual program. The teacher may keep a partially met description open and sample again.", "",
    ]
    names = {"science": "Science", "hass": "HASS", "hpe": "Health and Physical Education",
             "technologies": "Technologies", "arts": "The Arts"}
    for subject in SUBJECTS:
        lines += [f"## {names[subject]}", "", f"**Limit across these blocks:** {LIMIT[subject]}", "",
                  "| Code · official level | Original workbook row | Exact official content description | Day(s) with a concrete opportunity |",
                  "| --- | ---: | --- | --- |"]
        for code, row_no in sorted(ROWS[subject].items(), key=lambda pair: pair[1]):
            row = by_code[code]
            official = " ".join(row["plain_text"].split()).replace("|", "\\|")
            days = ", ".join(f"{day} ({card.title})" for day, card in enumerate(CARDS[subject], 11) if code in card.codes.split())
            check(days != "", f"{code}: unmapped")
            lines.append(f"| `{code}` · {LEVEL[subject]} | {row_no} | {official} | {days} |")
        lines += ["", ""]
    lines += [
        "## Read the alignment narrowly", "",
        "Science force investigation needs real safe trials; printed S-cards exercise data and source reasoning. HASS invented H-cards exercise inquiry skills without establishing place, history or civic facts. HPE permission and team codes need observed respectful action; movement codes need voluntary observed movement. Design testing/making codes need a prototype or actual process, and Digital implementation needs an actual learner-controlled Rule Lab run/change. Dance, Drama, Media Arts, Music and Visual Arts development, creation and presentation codes remain separate: a plan alone is never scored as performance, media production or sharing.", "",
        "[Staff evidence guides](README.md) identify block-specific limits and next moves. Any real local First Nations knowledge, Country/Place content, language, location or cultural form needs current authoritative source and appropriate community approval before use.", "",
        "**ACARA attribution:** © Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. Material downloaded from the [Australian Curriculum website](https://www.australiancurriculum.edu.au/downloads) (accessed 29 September 2026), modified only for plain-text whitespace, under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/) subject to [terms and exclusions](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse SubjectNest. Original SubjectNest crosswalk commentary © NeuroForgeIO Pty Ltd 2026, CC BY 4.0.", "",
    ]
    return "\n".join(lines)


def check_cards() -> None:
    check(tuple(CARDS) == SUBJECTS, "menu order/inventory drift")
    for subject, cards in CARDS.items():
        check(len(cards) == 10, f"{subject}: expected ten blocks")
        check(len({card.title for card in cards}) == 10, f"{subject}: titles not distinct")
        for card in cards:
            check(all(isinstance(v, str) and v.strip() for v in vars(card).values()), f"{subject}: missing field")
        for filename, expected in render(subject).items():
            path = ROOT / subject / filename
            check(path.read_text(encoding="utf-8") == expected, f"{path}: generated text drift")
        teacher = (ROOT / subject / "TEACHER.md").read_text(encoding="utf-8")
        learner = (ROOT / subject / "LEARNER.md").read_text(encoding="utf-8")
        staff = (ROOT / subject / "ASSESSMENT.md").read_text(encoding="utf-8")
        for day in range(11, 21):
            check(len(re.findall(rf"^## Day {day} — ", teacher, re.MULTILINE)) == 1, f"{subject}: Day {day}")
            check(len(re.findall(rf"^## Day {day} — ", learner, re.MULTILINE)) == 1, f"{subject}: learner Day {day}")
            check(len(re.findall(rf"^\| {day} \|", staff, re.MULTILINE)) == 1, f"{subject}: staff Day {day}")
        check(teacher.count("Notice · 2 min.") == teacher.count("Try/make · 7 min.") ==
              teacher.count("Show · 3 min.") == 10, f"{subject}: 2+7+3 format")
        check(teacher.count("**No-purchase home/low-material route:**") == 10, f"{subject}: home routes")
        check(learner.count("**Choose a route:**") == 10, f"{subject}: choices")
        check("**Evidence boundary:**" not in learner and "**Next teaching move:**" not in learner,
              f"{subject}: staff wording leaked to learner page")
    check(all(c.codes.startswith("AC9TDE") for c in CARDS["technologies"][:5]), "Design Days 11–15")
    check(all(c.codes.startswith("AC9TDI") for c in CARDS["technologies"][5:]), "Digital Days 16–20")
    for i, form in enumerate(("Dance", "Drama", "Media Arts", "Music", "Visual Arts")):
        check(all(c.title.startswith(form + ":") for c in CARDS["arts"][2*i:2*i+2]), f"{form}: two blocks")


def check_assets() -> None:
    alt = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    for subject in SUBJECTS:
        svg = ROOT / "print" / f"{subject}-aid.svg"
        pdf = svg.with_suffix(".pdf")
        check(svg.exists() and pdf.exists(), f"{subject}: missing A4 aid pair")
        source = svg.read_text(encoding="utf-8")
        check('width="210mm" height="297mm"' in source and 'role="img"' in source and
              "<title" in source and "<desc" in source and "CC BY 4.0" in source, f"{subject}: SVG access/size/rights")
        info = subprocess.run(["pdfinfo", str(pdf)], capture_output=True, text=True, check=True).stdout
        check(re.search(r"Pages:\s+1\b", info) is not None and "A4" in info, f"{subject}: PDF page geometry")
        extracted = subprocess.run(["pdftotext", str(pdf), "-"], capture_output=True, text=True, check=True).stdout
        check(len(extracted.strip()) > 150 and "SubjectNest" in extracted, f"{subject}: PDF text extraction")
        check(f"## {subject.title() if subject != 'hpe' else 'HPE'} aid" in alt or
              (subject == "arts" and "## Arts aid" in alt) or
              (subject == "hass" and "## HASS aid" in alt), f"{subject}: full text alternative")
    html = (ROOT / "interactive/rule-lab.html").read_text(encoding="utf-8")
    check(all(f'id="{id_}"' in html for id_ in ("match", "yes", "no", "token1", "token2", "token3", "run", "reset", "summary", "trace")),
          "Rule Lab controls")
    check("<script src=" not in html and "localStorage" not in html and "fetch(" not in html,
          "Rule Lab must be self-contained and not store/send data")
    subprocess.run(["node", str(ROOT / "interactive/test_rule_lab.mjs")], check=True)


def slug(heading: str) -> str:
    heading = re.sub(r"\[[^]]+\]\([^)]*\)", "", heading)
    return re.sub(r"[^a-z0-9-]", "", heading.strip().lower().replace(" ", "-"))


def check_links() -> int:
    checked = 0
    for path in ROOT.rglob("*.md"):
        source = path.read_text(encoding="utf-8")
        check("CC BY 4.0" in source, f"{path}: missing licence notice")
        for raw in re.findall(r"\[[^]]+\]\(([^)]+)\)", source):
            if re.match(r"https?://", raw):
                continue
            relative, _, anchor = raw.partition("#")
            target = (path.parent / relative).resolve() if relative else path
            check(target.exists(), f"{path}: broken local link {raw}")
            if anchor and target.suffix.lower() == ".md":
                headings = [slug(match.group(1)) for match in re.finditer(r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)]
                check(anchor in headings, f"{path}: missing heading #{anchor} in {target}")
            checked += 1
    return checked


def payload_files() -> list[Path]:
    return sorted((p for p in ROOT.rglob("*") if p.is_file() and p != MANIFEST and
                   "__pycache__" not in p.parts and ".ruff_cache" not in p.parts),
                  key=lambda p: p.relative_to(ROOT).as_posix())


def manifest_text() -> str:
    return "".join(f"{digest(p)}  {p.relative_to(ROOT).as_posix()}\n" for p in payload_files())


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    check_cards()
    by_code = framework_rows()
    target = ROOT / "CURRICULUM-CROSSWALK.md"
    expected = crosswalk(by_code)
    if args.write_crosswalk:
        target.write_text(expected, encoding="utf-8")
    else:
        check(target.read_text(encoding="utf-8") == expected, "crosswalk drift from pinned descriptions")
    check_assets()
    links = check_links()
    expected_manifest = manifest_text()
    if args.write_manifest:
        MANIFEST.write_text(expected_manifest, encoding="utf-8")
    else:
        check(MANIFEST.read_text(encoding="utf-8") == expected_manifest, "SHA manifest drift")
    print(f"PASS: 5 menus × 10 days × 12 min; 50 distinct blocks; 51 exact codes/rows; "
          f"15 teacher/learner/staff pages; 5 A4 SVG/PDF/text aids; {links} local links; "
          f"{len(payload_files())} hashed files")


if __name__ == "__main__":
    main()
