#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Audit the original Year 4 English continuation and its pinned curriculum source."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
IMPORT = STUDIO / "data/frameworks/acara-v9.json"
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = ["AC9E4LA02", "AC9E4LA03", "AC9E4LA04", "AC9E4LA05", "AC9E4LA06", "AC9E4LA10", "AC9E4LA11", "AC9E4LY02", "AC9E4LY03", "AC9E4LY04", "AC9E4LY05", "AC9E4LY06", "AC9E4LY07", "AC9E4LY09", "AC9E4LY10"]
EVIDENCE = {
    "AC9E4LA02": "Days 13, 17 and 20 sort source-supported report from a marked preference; no whole-year opinion-language claim.",
    "AC9E4LA03": "Days 11 and 13 compare report, procedure and story stages; Check A transfers structure to a new page.",
    "AC9E4LA04": "Days 14 and 18 use temporal, conditional and causal links in short source-bound sentences.",
    "AC9E4LA05": "Day 12 examines a text-only online mockup's headline, breadcrumb, menu, link and headings; Check A is a new mockup, not a tested site.",
    "AC9E4LA06": "Days 14 and 18 join a complete clause to a dependent because/if clause; a brief application only.",
    "AC9E4LA10": "Day 16 contrasts top and side framing; Check B asks what a side view foregrounds, not full moving-image study.",
    "AC9E4LA11": "Day 17 tests whether a synonym or antonym preserves a source claim; limited word set.",
    "AC9E4LY02": "Days 13 and 19 ask learners to acknowledge a listener's interpretation and respond to a question, in flexible modes.",
    "AC9E4LY03": "Days 11 and 13 identify purposes and features of several text types; Check A transfers to a new page.",
    "AC9E4LY04": "Days 11, 12 and 16 include separate matched independent reading; adult-read packet comprehension is not fluency evidence.",
    "AC9E4LY05": "Across both weeks pupils locate, question, summarise and evaluate model limits; fresh checks sample transfer.",
    "AC9E4LY06": "Days 14 and 18–20 plan, compose, revise and check short informative text; other genres need later work.",
    "AC9E4LY07": "Day 19 rehearses one brief audience-specific oral/AAC/multimodal explanation and revises after a question.",
    "AC9E4LY09": "Days 14 and 17 practise known morphemes and word parts; two fresh checks sample new forms after taught-prerequisite review.",
    "AC9E4LY10": "Days 14 and 17 examine word-family spelling; fresh words invite explanation without asserting full spelling mastery.",
}
HELD_OUT = ("relabelled", "unmarked", "repositioned", "unmeasured")


def require(test: bool, message: str) -> None:
    if not test:
        raise AssertionError(message)


def norm(value: str) -> str:
    return " ".join(value.split())


def source() -> tuple[dict, dict[str, dict]]:
    data = json.loads(IMPORT.read_text(encoding="utf-8"))
    require(data["source_sha256"] == SOURCE_SHA, "Pinned ACARA source SHA changed")
    original = STUDIO / "research/sources" / data["source_file"]
    require(original.is_file(), "Official workbook bytes missing")
    require(hashlib.sha256(original.read_bytes()).hexdigest() == SOURCE_SHA, "Official workbook bytes changed")
    require(data["framework"] == "Australian Curriculum Version 9.0", "Framework version changed")
    require(data["source_url"].startswith("https://www.australiancurriculum.edu.au/"), "Nonofficial source URL")
    records = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description" and r.get("code")}
    for code in CODES:
        require(code in records, f"Missing source code {code}")
        attrs = records[code]["attributes"]
        require(attrs["level"] == "Year 4" and attrs["learning_area"] == "English" and attrs["subject"] == "English", f"Wrong level or subject for {code}")
        require(isinstance(records[code]["source_row"], int), f"Missing source row {code}")
    return data, records


def make_crosswalk(data: dict, records: dict[str, dict]) -> str:
    lines = [
        "# Australian Curriculum v9 · Year 4 English Weeks 3–4 crosswalk", "",
        "These are **partial links to ten 25-minute sessions**, not proof of full content-description coverage, Year 4 achievement, a mandated order or state/territory adoption. Adult-read comprehension is not independent reading evidence.", "",
        f"Official source: [ACARA curriculum workbook]({data['source_url']}), retrieved {data['retrieved_at']}; original XLSX SHA-256 `{data['source_sha256']}`. Exact description wording follows the source with whitespace normalised; source rows permit an audit.", "",
        "| Official code | Workbook row | Level | Exact official content description | Taught opportunity and limit |",
        "|---|---:|---|---|---|",
    ]
    for code in CODES:
        rec = records[code]
        cells = (code, str(rec["source_row"]), rec["attributes"]["level"], norm(rec["plain_text"]), EVIDENCE[code])
        lines.append("| " + " | ".join(c.replace("|", "\\|") for c in cells) + " |")
    lines += [
        "", "© Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. Curriculum wording was downloaded from the [Australian Curriculum website](https://www.australiancurriculum.edu.au/downloads) (accessed 29 September 2026), with whitespace/plain-text normalisation, under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/) subject to [ACARA terms and exclusions](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse SubjectNest.",
        "", "Original evidence notes and table arrangement © NeuroForgeIO Pty Ltd 2026, SubjectNest, CC BY 4.0. See the [source ledger](SOURCES-AND-REVIEW.md). Recheck current official and local sources before reissue.", "",
    ]
    return "\n".join(lines)


def slug(heading: str) -> str:
    return "".join(c for c in heading.lower() if c.isalnum() or c in " -_").replace(" ", "-")


def links_rights() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        body = path.read_text(encoding="utf-8")
        require("CC BY 4.0" in body, f"Rights missing: {path.relative_to(ROOT)}")
        for target in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", body):
            if target.startswith(("https://", "http://", "mailto:")):
                continue
            part, _, anchor = unquote(target).partition("#")
            dest = (path.parent / part).resolve() if part else path
            require(dest.is_file(), f"Broken local link {path.relative_to(ROOT)} -> {target}")
            if anchor and dest.suffix == ".md":
                other = dest.read_text(encoding="utf-8")
                heads = re.findall(r"^#{1,6}\s+(.+?)\s*$", other, re.MULTILINE)
                ids = set(re.findall(r'<a\s+id="([^"]+)"\s*></a>', other))
                require(anchor in {slug(h) for h in heads} | ids, f"Broken anchor {path.relative_to(ROOT)} -> {target}")
            count += 1
    return count


def content() -> None:
    scripts = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    learner = (ROOT / "LEARNER.md").read_text(encoding="utf-8")
    extras = (ROOT / "DAILY-EXTRAS.md").read_text(encoding="utf-8")
    checks = (ROOT / "STUDENT-CHECKS.md").read_text(encoding="utf-8")
    key = (ROOT / "teacher/ANSWER-AND-NEXT.md").read_text(encoding="utf-8")
    for name, body in (("LESSONS", scripts), ("LEARNER", learner), ("DAILY-EXTRAS", extras)):
        days = [int(d) for d in re.findall(r"^## Day (\d+)\b", body, re.MULTILINE)]
        require(days == list(range(11, 21)), f"{name}: expected Days 11–20 once, got {days}")
    lesson_blocks = re.split(r"^## Day \d+\b", scripts, flags=re.MULTILINE)[1:]
    for day, block in enumerate(lesson_blocks, 11):
        minutes = [int(x) for x in re.findall(r"^\d+\. \*\*[^*]+ · (\d+) min\.\*\*", block, re.MULTILINE)]
        expected = [2, 3, 3, 12, 3, 2] if day in (15, 20) else [2, 5, 6, 7, 3, 2]
        require(minutes == expected, f"Day {day} phases drift: {minutes}")
        require("AC9E4" in block, f"Day {day} missing code")
        require(f"Day {day}" in learner and f"Day {day}" in extras, f"Day {day} choices missing")
    require(set(re.findall(r"\bAC9E4[A-Z0-9]+\b", scripts)) == set(CODES), "Script code set differs from exact crosswalk")
    require(re.findall(r"^- \*\*([ABC]) · ", learner, re.MULTILINE) == list("ABC") * 10, "Exactly 30 core choice cards required")
    require(re.findall(r"^- \*\*([AB]) · ", extras, re.MULTILINE) == list("AB") * 10, "Exactly 20 optional choice cards required")
    for word in HELD_OUT:
        for name in ("LESSONS.md", "LEARNER.md", "DAILY-EXTRAS.md", "MATERIALS.md", "print/TEXT-ALTERNATIVES.md"):
            require(re.search(rf"\b{word}\b", (ROOT / name).read_text(encoding="utf-8"), re.IGNORECASE) is None, f"Held-out word {word} leaked into {name}")
        require(re.search(rf"\b{word}\b", checks, re.IGNORECASE) is not None and re.search(rf"\b{word}\b", key, re.IGNORECASE) is not None, f"Held-out word {word} absent from check/key")
    require("adult-read texts, not controlled decodables" in (ROOT / "README.md").read_text(encoding="utf-8").lower(), "Independent-reading boundary missing")
    require("not yet observed" in key.lower(), "Missing evidence handling absent")
    for term in ("Check A", "Check B", "Day 15", "Day 20"):
        require(term in checks and term in key, f"Fresh check/key incomplete: {term}")
    for term in ("two trays", "one handle", "RETURN", "three panels", "two square picture cards", "no dimensions"):
        require(term.lower() in checks.lower(), f"Fresh model missing detail {term}")


def visuals() -> None:
    ns = "{http://www.w3.org/2000/svg}"
    for stem, phrases in {
        "source-purpose-sorter": ("One plan, four text jobs", "REPORT", "PROCEDURE", "VIEWPOINT", "STORY", "One thing the source cannot prove"),
        "model-evidence-planner": ("Model", "WHAT THE MODEL SHOWS", "WHAT SOMEONE THINKS", "WHAT IS STILL UNKNOWN", "My short summary"),
    }.items():
        svg = ROOT / "print" / f"{stem}.svg"
        pdf = svg.with_suffix(".pdf")
        tree = ET.parse(svg).getroot()
        require(tree.get("viewBox") == "0 0 210 297" and tree.get("role") == "img", f"{stem}: A4/access metadata drift")
        require(tree.find(ns + "title") is not None and tree.find(ns + "desc") is not None, f"{stem}: title/desc missing")
        require("CC BY 4.0" in svg.read_text(encoding="utf-8"), f"{stem}: rights missing")
        info = subprocess.run(["pdfinfo", str(pdf)], capture_output=True, text=True, check=True).stdout
        require("(A4)" in info and "Pages:           1" in info, f"{stem}: PDF not one A4 page")
        pdf_text = subprocess.run(["pdftotext", str(pdf), "-"], capture_output=True, text=True, check=True).stdout
        for phrase in phrases:
            require(phrase in pdf_text, f"{stem}: PDF text missing {phrase}")
        fonts = subprocess.run(["pdffonts", str(pdf)], capture_output=True, text=True, check=True).stdout
        require("DejaVu" in fonts, f"{stem}: embedded font changed")
    alt = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    for phrase in ("REPORT", "PROCEDURE", "VIEWPOINT", "STORY", "WHAT THE MODEL SHOWS", "WHAT IS STILL UNKNOWN", "Tactile build"):
        require(phrase in alt, f"Text/tactile alternative missing {phrase}")


def hashes() -> str:
    paths = sorted(p for p in ROOT.rglob("*") if p.is_file() and p.name != "MANIFEST.sha256" and "__pycache__" not in p.parts)
    return "".join(f"{hashlib.sha256(p.read_bytes()).hexdigest()}  {p.relative_to(ROOT)}\n" for p in paths)


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    data, records = source()
    expected = make_crosswalk(data, records)
    path = ROOT / "CURRICULUM-CROSSWALK.md"
    if args.write_crosswalk:
        path.write_text(expected, encoding="utf-8")
    require(path.read_text(encoding="utf-8") == expected, "Crosswalk differs from pinned workbook")
    content()
    visuals()
    count = links_rights()
    digest = hashes()
    manifest = ROOT / "MANIFEST.sha256"
    if args.write_manifest:
        manifest.write_text(digest, encoding="utf-8")
    require(manifest.read_text(encoding="utf-8") == digest, "Hash manifest drift")
    print(f"PASS: 10×25-minute days; 30 core and 20 optional routes; 2 held-out checks; {len(CODES)} exact Year 4 English codes; 2 original A4 SVG/PDF/text aids; {count} local links; {len(digest.splitlines())} hashes")


if __name__ == "__main__":
    try:
        main()
    except (AssertionError, FileNotFoundError, KeyError, subprocess.CalledProcessError) as exc:
        print(f"FAIL: {exc}", file=sys.stderr)
        raise SystemExit(1)
