#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only audit of the Year 3 English continuation; writing needs explicit flags."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
IMPORT = STUDIO / "data/frameworks/acara-v9.json"
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = "AC9E3LA03 AC9E3LA04 AC9E3LA05 AC9E3LA07 AC9E3LA08 AC9E3LA09 AC9E3LA10 AC9E3LE03 AC9E3LY01 AC9E3LY02 AC9E3LY03 AC9E3LY05 AC9E3LY06 AC9E3LY07 AC9E3LY09 AC9E3LY10".split()
EVIDENCE = {
    "AC9E3LA03": "Days 13 and 16 compare an instruction with an explanation; not a full survey of curriculum texts.",
    "AC9E3LA04": "Day 18 groups starting label and change/check into related idea paragraphs; a brief composition sample.",
    "AC9E3LA05": "Days 13 and 15 use headings/labels to find or repair information; print focus, no digital navigation claim.",
    "AC9E3LA07": "Days 12 and 17 compare doing and relating verbs in route/explanation sentences; other verb processes need later teaching.",
    "AC9E3LA08": "Day 17 contrasts past report and present instruction; short tense encounter.",
    "AC9E3LA09": "Day 18 checks whether a labelled diagram extends meaning and agrees with the words; one visual form only.",
    "AC9E3LA10": "Days 12 and 16 clarify route and stock-card words, including exchange in this context; small vocabulary sample.",
    "AC9E3LE03": "Day 11 discusses the story's feeling clue and setting diagram; limited language/visual discussion, not a literature range.",
    "AC9E3LY01": "Days 14 and 19 revise for a second reader and an absent classmate; familiar audiences only.",
    "AC9E3LY02": "Days 11, 14 and 19 make space for questions and responsive talk, including AAC; no formal interaction judgement.",
    "AC9E3LY03": "Days 13 and 20 name information-text purpose and the visitor audience; not all persuasive/imaginative types.",
    "AC9E3LY05": "Days 11, 13, 16 and fresh checks use literal/inferred meaning after adult read-aloud; never scored as independent decoding.",
    "AC9E3LY06": "Days 12–15 edit short directions; Days 17–20 plan, compose and revise a short explanation; no full genre coverage.",
    "AC9E3LY07": "Day 19 rehearses and gives a brief audience-specific oral/AAC or multimodal explanation; one occasion only.",
    "AC9E3LY09": "Day 14 applies taught letter/syllable patterns to longer moving/making and fresh transfer where appropriate; teacher audits prerequisites.",
    "AC9E3LY10": "Daily prefix/suffix/base work and fresh Day 15/20 words sample application; generalisations need wider teaching.",
}
HELD_OUT_WORDS = ("refill", "painter", "shaping", "rebuild", "builder", "tracing")


def require(value: bool, message: str) -> None:
    if not value:
        raise AssertionError(message)


def norm(value: str) -> str:
    return " ".join(value.split())


def source() -> tuple[dict, dict[str, dict]]:
    data = json.loads(IMPORT.read_text(encoding="utf-8"))
    require(data["source_sha256"] == SOURCE_SHA, "Pinned source SHA drift")
    workbook = STUDIO / "research/sources" / data["source_file"]
    require(workbook.is_file(), "Official workbook missing")
    require(hashlib.sha256(workbook.read_bytes()).hexdigest() == SOURCE_SHA, "Official workbook bytes drift")
    require(data["framework"] == "Australian Curriculum Version 9.0", "Framework drift")
    require(data["source_url"].startswith("https://www.australiancurriculum.edu.au/"), "Nonofficial source URL")
    records = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description" and r.get("code")}
    for code in CODES:
        require(code in records, f"Unknown curriculum code {code}")
        attrs = records[code]["attributes"]
        require(attrs["learning_area"] == "English" and attrs["subject"] == "English" and attrs["level"] == "Year 3", f"Wrong level/subject for {code}")
    return data, records


def crosswalk(data: dict, records: dict[str, dict]) -> str:
    lines = [
        "# Australian Curriculum v9 · Year 3 English Weeks 3–4 crosswalk", "",
        "These are **partial links to ten 25-minute sessions**. They do not imply full content-description coverage, an achievement-standard decision, a prescribed teaching order or automatic state/territory adoption. Adult-read comprehension is not independent reading evidence.", "",
        f"Official source: [ACARA curriculum workbook]({data['source_url']}), retrieved {data['retrieved_at']}; original XLSX SHA-256 `{data['source_sha256']}`. Description wording is exact apart from whitespace normalisation. Workbook source rows permit an audit.", "",
        "| Official code | Workbook row | Level | Exact official content description | Taught opportunity and limit |",
        "|---|---:|---|---|---|",
    ]
    for code in CODES:
        record = records[code]
        cells = (code, str(record["source_row"]), record["attributes"]["level"], norm(record["plain_text"]), EVIDENCE[code])
        lines.append("| " + " | ".join(c.replace("|", "\\|") for c in cells) + " |")
    lines += [
        "", "© Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. This material was downloaded from the [Australian Curriculum website](https://www.australiancurriculum.edu.au/downloads) (accessed 29 September 2026) and was modified by whitespace/plain-text normalisation. It is licensed under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/). [ACARA terms and exclusions](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse, sponsor or approve SubjectNest.",
        "", "Original evidence descriptions and table arrangement © NeuroForgeIO Pty Ltd 2026, SubjectNest, CC BY 4.0. See the [source and review ledger](SOURCES-AND-REVIEW.md). Recheck the official version and the applicable local syllabus before a future edition.", "",
    ]
    return "\n".join(lines)


def slug(heading: str) -> str:
    return "".join(c for c in heading.lower() if c.isalnum() or c in " -_").replace(" ", "-")


def links_and_rights() -> int:
    n = 0
    for path in ROOT.rglob("*.md"):
        body = path.read_text(encoding="utf-8")
        require("CC BY 4.0" in body, f"Rights missing in {path.relative_to(ROOT)}")
        for target in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", body):
            if target.startswith(("https://", "http://", "mailto:")):
                continue
            file_part, _, anchor = unquote(target).partition("#")
            destination = (path.parent / file_part).resolve() if file_part else path
            require(destination.is_file(), f"Broken link {path.relative_to(ROOT)} -> {target}")
            if anchor and destination.suffix == ".md":
                heads = re.findall(r"^#{1,6}\s+(.+?)\s*$", destination.read_text(encoding="utf-8"), re.M)
                require(anchor in {slug(h) for h in heads}, f"Broken heading anchor {path.relative_to(ROOT)} -> {target}")
            n += 1
    return n


def content() -> None:
    lessons = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    learner = (ROOT / "LEARNER.md").read_text(encoding="utf-8")
    extras = (ROOT / "DAILY-EXTRAS.md").read_text(encoding="utf-8")
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8")
    student = (ROOT / "STUDENT-CHECKS.md").read_text(encoding="utf-8")
    key = (ROOT / "teacher/ANSWER-AND-NEXT.md").read_text(encoding="utf-8")
    for label, body in (("LESSONS", lessons), ("LEARNER", learner)):
        days = [int(d) for d in re.findall(r"^## Day (\d+)\b", body, re.M)]
        require(days == list(range(11, 21)), f"{label}: Days 11–20 expected once: {days}")
    blocks = re.split(r"^## Day \d+\b", lessons, flags=re.M)[1:]
    for day, block in enumerate(blocks, 11):
        marks = re.findall(r"^\d+\. \*\*[^*]+ · (\d+) min\.\*\*", block, re.M)
        require(marks == ["2", "5", "6", "7", "3", "2"], f"Day {day}: 25-minute phases drift: {marks}")
        require("AC9E3" in block, f"Day {day}: no exact code link")
        require(f"Day {day}" in learner, f"Day {day}: no learner work")
    choices = re.findall(r"^- \*\*([ABC]) · ", learner, re.M)
    require(choices == list("ABC") * 10, "Exactly A/B/C routes required each day")
    extra_days = [int(d) for d in re.findall(r"^## Day (\d+)\b", extras, re.M)]
    require(extra_days == list(range(11, 21)), f"Daily extras missing a day: {extra_days}")
    require(re.findall(r"^- \*\*([AB]) · ", extras, re.M) == list("AB") * 10, "Exactly A/B extras required each day")
    require(set(re.findall(r"\bAC9E3[A-Z0-9]+\b", lessons)) == set(CODES), "Lesson code set drift")
    for label, body in (("LESSONS", lessons), ("LEARNER", learner), ("DAILY-EXTRAS", extras), ("MATERIALS", materials), ("print alt", (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8"))):
        for word in HELD_OUT_WORDS:
            require(re.search(rf"\b{word}\b", body, re.I) is None, f"Held-out word {word} leaked into {label}")
        require("13 204" not in body, f"Held-out count leaked into {label}")
    for word in HELD_OUT_WORDS:
        require(re.search(rf"\b{word}\b", student, re.I) is not None, f"Held-out word absent in check: {word}")
        require(re.search(rf"\b{word}\b", key, re.I) is not None, f"Held-out word absent in key: {word}")
    for day, label in ((15, "Check A"), (20, "Check B")):
        require(f"Day {day}" in student and label in student, f"Day {day} fresh check missing")
        require(f"Day {day}" in key and label in key, f"Day {day} teacher key missing")
    require("not controlled decodables" in materials.lower(), "Shared/independent reading boundary missing")
    require("not yet observed" in key.lower(), "Missing-evidence handling absent")
    require(20_000 + 4_000 + 300 + 6 == 10_000 + 14_000 + 300 + 6 == 24_306, "Practice regroup arithmetic false")
    require(300 + 40 == 200 + 140 == 340, "Small model arithmetic false")
    require(1_000 + 200 + 30 == 1_200 + 30 == 1_230, "Optional small exchange arithmetic false")
    require(10_000 + 3_000 + 200 + 4 == 13_000 + 200 + 4 == 13_204, "Fresh check arithmetic false")
    for phrase in ("DOT", "BELL", "STAR", "FLAG", "two squares up", "two squares right"):
        require(phrase in student, f"Route check incomplete: {phrase}")


def visuals() -> None:
    for stem, phrases in {
        "garden-route-mat": ("A route another reader can test", "ENTRANCE", "NOTICE BOARD", "SHADE TREE", "Reader paused at:"),
        "stock-explanation-planner": ("One total", "FIRST LABEL", "AFTER THE EXCHANGE", "What changed?", "How do you know?"),
    }.items():
        svg = ROOT / "print" / f"{stem}.svg"
        pdf = svg.with_suffix(".pdf")
        elem = ET.parse(svg).getroot()
        ns = "{http://www.w3.org/2000/svg}"
        require(elem.get("viewBox") == "0 0 210 297" and elem.get("role") == "img", f"{stem}: A4/access metadata drift")
        require(elem.find(ns + "title") is not None and elem.find(ns + "desc") is not None, f"{stem}: title/description missing")
        require("CC BY 4.0" in svg.read_text(encoding="utf-8"), f"{stem}: rights missing")
        info = subprocess.run(["pdfinfo", str(pdf)], capture_output=True, text=True, check=True).stdout
        require("(A4)" in info and "Pages:           1" in info, f"{stem}: PDF not one A4 page")
        extracted = subprocess.run(["pdftotext", str(pdf), "-"], capture_output=True, text=True, check=True).stdout
        for phrase in phrases:
            require(phrase in extracted, f"{stem}: PDF text missing {phrase}")
        fonts = subprocess.run(["pdffonts", str(pdf)], capture_output=True, text=True, check=True).stdout
        require("DejaVu" in fonts, f"{stem}: font embed changed")
    alt = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    for term in ("ENTRANCE", "NOTICE BOARD", "SHADE TREE", "top row", "What changed?", "What stayed the same?", "Tactile build"):
        require(term in alt, f"Text/tactile alternative missing {term}")


def hashes() -> str:
    paths = sorted(p for p in ROOT.rglob("*") if p.is_file() and p.name != "MANIFEST.sha256" and "__pycache__" not in p.parts)
    return "".join(f"{hashlib.sha256(p.read_bytes()).hexdigest()}  {p.relative_to(ROOT)}\n" for p in paths)


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    data, records = source()
    expected = crosswalk(data, records)
    crosswalk_path = ROOT / "CURRICULUM-CROSSWALK.md"
    if args.write_crosswalk:
        crosswalk_path.write_text(expected, encoding="utf-8")
    require(crosswalk_path.read_text(encoding="utf-8") == expected, "Pinned ACARA crosswalk drift")
    content()
    visuals()
    nlinks = links_and_rights()
    digest = hashes()
    manifest = ROOT / "MANIFEST.sha256"
    if args.write_manifest:
        manifest.write_text(digest, encoding="utf-8")
    require(manifest.read_text(encoding="utf-8") == digest, "Hash manifest drift")
    print(f"PASS: 10×25-minute days · 30 core + 20 optional learner routes · 2 fresh checks · {len(CODES)} exact Year 3 English links · 2 original A4 SVG/PDF/text aids · {nlinks} links · {len(digest.splitlines())} hashes")


if __name__ == "__main__":
    try:
        main()
    except (AssertionError, FileNotFoundError, KeyError, subprocess.CalledProcessError) as error:
        print(f"FAIL: {error}", file=sys.stderr)
        raise SystemExit(1)
