#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only audit by default; explicit flags update derived crosswalk/hashes."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
IMPORT = STUDIO / "data/frameworks/acara-v9.json"
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = "AC9E2LA03 AC9E2LA05 AC9E2LA06 AC9E2LA09 AC9E2LE02 AC9E2LY02 AC9E2LY03 AC9E2LY05 AC9E2LY06 AC9E2LY07 AC9E2LY10".split()
EVIDENCE = {
    "AC9E2LA03": "Days 13, 16 and 20 compare story/how-to/explanation structure; only a small sample of text purposes.",
    "AC9E2LA05": "Day 13 uses headings to locate a useful step; chapters, menus and indexes are not taught here.",
    "AC9E2LA06": "Days 17–18 orally compose and edit linked independent ideas; a short starting opportunity.",
    "AC9E2LA09": "Days 12 and 18 compare a vague verb/reason with more precise topic words.",
    "AC9E2LE02": "Day 11 names characters, setting and a preference with a reason after adult-read literature.",
    "AC9E2LY02": "Paper-token partner directions and attentive revision occur on Days 11–15 and 19.",
    "AC9E2LY03": "Day 13 identifies an information text's purpose and partner audience.",
    "AC9E2LY05": "Prediction, questioning, locating and monitoring meaning occur after adult read-aloud; not independent text decoding evidence.",
    "AC9E2LY06": "Days 12–15 compose/revise short directions; Days 18–20 compose an explanation. This samples the description, not all genres/features.",
    "AC9E2LY07": "Day 19 rehearses and delivers a short explanation with pace/pointing for a familiar listener; supported modes recorded.",
    "AC9E2LY10": "Daily locally audited vowel-digraph word reading/writing; no universal sequence and no full range of listed patterns.",
}
HELD_OUT = ("mail", "hay", "peel", "meal", "tail", "bay", "feed", "seat")


def require(value: bool, message: str) -> None:
    if not value:
        raise AssertionError(message)


def norm(value: str) -> str:
    return " ".join(value.split())


def official() -> tuple[dict, dict[str, dict]]:
    data = json.loads(IMPORT.read_text(encoding="utf-8"))
    require(data["source_sha256"] == SOURCE_SHA, "Pinned source digest changed")
    source = STUDIO / "research/sources" / data["source_file"]
    require(source.is_file(), f"Pinned workbook missing: {source}")
    require(hashlib.sha256(source.read_bytes()).hexdigest() == SOURCE_SHA, "Pinned workbook bytes changed")
    require(data["framework"] == "Australian Curriculum Version 9.0", "Framework changed")
    require(data["source_url"].startswith("https://www.australiancurriculum.edu.au/"), "Source URL changed")
    records = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description"}
    for code in CODES:
        require(code in records, f"Unknown code {code}")
        a = records[code]["attributes"]
        require(a["learning_area"] == "English" and a["subject"] == "English" and a["level"] == "Year 2", f"Wrong level/area: {code}")
    return data, records


def crosswalk(data: dict, records: dict[str, dict]) -> str:
    lines = [
        "# Australian Curriculum v9 · Year 2 English Weeks 3–4 crosswalk", "",
        "These are **partial links to ten 25-minute lessons**, not full content-description coverage, an achievement-standard judgment or a school-wide spelling order. Adult-read comprehension is not independent text-reading evidence.", "",
        f"Official source: [ACARA curriculum workbook]({data['source_url']}), retrieved {data['retrieved_at']}; workbook SHA-256 `{data['source_sha256']}`. Wording is exact apart from whitespace normalisation. The workbook source row is retained for audit.", "",
        "| Official code | Workbook row | Official level | Official content description | Evidence in this block and limit |",
        "|---|---:|---|---|---|",
    ]
    for code in CODES:
        r = records[code]
        cells = (code, str(r["source_row"]), r["attributes"]["level"], norm(r["plain_text"]), EVIDENCE[code])
        lines.append("| " + " | ".join(s.replace("|", "\\|") for s in cells) + " |")
    lines += ["", "© Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. Downloaded from the Australian Curriculum website (accessed 29 September 2026) and modified for plain-text display. Curriculum material is licensed under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/). [Terms and exclusions](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse or approve SubjectNest.", "", "The import is a dated snapshot. Recheck the current official source and relevant state/territory implementation for future reissue. See [source and review ledger](SOURCES-AND-REVIEW.md).", ""]
    return "\n".join(lines)


def check_content() -> None:
    lessons = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    learner = (ROOT / "LEARNER.md").read_text(encoding="utf-8")
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8")
    assessment = (ROOT / "teacher/ASSESSMENT.md").read_text(encoding="utf-8")
    for label, doc in (("lessons", lessons), ("learner", learner)):
        days = [int(x) for x in re.findall(r"^## Day (\d+)\b", doc, re.M)]
        require(days == list(range(11, 21)), f"{label}: Days 11–20 expected once each")
    sections = re.split(r"^## Day \d+\b", lessons, flags=re.M)[1:]
    for day, section in enumerate(sections, 11):
        for phase in ("**2 retrieve:**", "**5 model:**", "**6 code:**", "**7 choice:**", "**3 explain/revise:**", "**2 exit:**"):
            require(phase in section, f"Day {day}: missing {phase}")
        require("AC9E2" in section, f"Day {day}: missing curriculum link")
    declared = set(re.findall(r"\bAC9E2[A-Z0-9]+\b", lessons))
    require(declared == set(CODES), f"Lesson code set mismatch: {declared ^ set(CODES)}")
    cards = re.split(r"^## Day \d+\b", learner, flags=re.M)[1:]
    for day, card in enumerate(cards, 11):
        for marker in ("**Read:**", "**Write:**", "**Optional short sentence:**", "**Finish:**"):
            require(marker in card, f"Day {day} learner card missing {marker}")
        for letter in ("A", "B", "C"):
            require(re.search(rf"^- \*\*{letter} · ", card, re.M) is not None, f"Day {day} missing option {letter}")
        require(len(re.findall(r"^- \*\*[ABC] · ", card, re.M)) == 3, f"Day {day}: exactly three options required")
    require("adult reads" in materials.lower() or "adult-read" in materials.lower(), "Adult-read boundary missing")
    require("not" in materials.lower() and "decodable" in materials.lower(), "Decodability boundary missing")
    for word in HELD_OUT:
        for name, doc in (("LESSONS", lessons), ("LEARNER", learner), ("MATERIALS", materials)):
            require(re.search(rf"\b{word}\b", doc, re.I) is None, f"Held-out word {word} leaked into {name}")
        require(re.search(rf"\b{word}\b", assessment, re.I) is not None, f"Held-out word missing: {word}")
    for day in (15, 20):
        require(f"## Day {day}" in assessment, f"Missing Day {day} check")
    for term in ("Exact key", "Next moves", "24", "STAR → CIRCLE → SQUARE → BOOK", "not yet taught"):
        require(term.lower() in assessment.lower(), f"Assessment missing {term}")
    require("mail" not in (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8"), "Print leaked answer")


def check_links() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        doc = path.read_text(encoding="utf-8")
        for target in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", doc):
            if target.startswith(("https://", "http://", "mailto:", "#")):
                continue
            destination = (path.parent / unquote(target.split("#", 1)[0])).resolve()
            require(destination.is_file(), f"Broken local link: {path.relative_to(ROOT)} -> {target}")
            count += 1
    return count


def check_asset() -> None:
    svg = ROOT / "print/route-and-reason.svg"
    pdf = ROOT / "print/route-and-reason.pdf"
    image = ET.parse(svg).getroot()
    ns = {"s": "http://www.w3.org/2000/svg"}
    require(image.attrib["width"] == "210mm" and image.attrib["height"] == "297mm", "SVG not A4")
    require(image.find("s:title", ns) is not None and image.find("s:desc", ns) is not None, "SVG access text missing")
    require(pdf.is_file(), "PDF missing")
    info = subprocess.run(["pdfinfo", str(pdf)], check=True, capture_output=True, text=True).stdout
    require("(A4)" in info and "Pages:           1" in info, "PDF is not one A4 page")
    extracted = subprocess.run(["pdftotext", str(pdf), "-"], check=True, capture_output=True, text=True).stdout
    for phrase in ("Route and reason mat", "A reader paused at", "What changed?", "CC BY 4.0"):
        require(phrase in extracted, f"PDF text missing {phrase}")
    fonts = subprocess.run(["pdffonts", str(pdf)], check=True, capture_output=True, text=True).stdout
    require("DejaVu" in fonts, "Embedded DejaVu subset expected")
    alt = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    for phrase in ("Start:", "Landmark / turn:", "End:", "What changed?", "Why did the whole"):
        require(phrase in alt, f"Text alternative missing {phrase}")


def hashes() -> str:
    paths = sorted(p for p in ROOT.rglob("*") if p.is_file() and p.name != "MANIFEST.sha256" and "__pycache__" not in p.parts)
    return "".join(f"{hashlib.sha256(p.read_bytes()).hexdigest()}  {p.relative_to(ROOT)}\n" for p in paths)


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    data, records = official()
    content = crosswalk(data, records)
    path = ROOT / "CURRICULUM-CROSSWALK.md"
    if args.write_crosswalk:
        path.write_text(content, encoding="utf-8")
    require(path.read_text(encoding="utf-8") == content, "Crosswalk drift from pinned ACARA import")
    check_content()
    links = check_links()
    check_asset()
    digest = hashes()
    manifest = ROOT / "MANIFEST.sha256"
    if args.write_manifest:
        manifest.write_text(digest, encoding="utf-8")
    require(manifest.read_text(encoding="utf-8") == digest, "Hash manifest drift")
    print(f"PASS: 10 lessons · 10 learner cards · 30 choice routes · 2 held-out checks · {len(CODES)} exact ACARA codes · A4 SVG/PDF and text aid · {links} local links · {len(digest.splitlines())} hashes")


if __name__ == "__main__":
    try:
        main()
    except (AssertionError, KeyError, FileNotFoundError, subprocess.CalledProcessError) as exc:
        print(f"FAIL: {exc}", file=sys.stderr)
        raise SystemExit(1)
