#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only validation; optional generation of exact ACARA crosswalk/manifest."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
IMPORT = STUDIO / "data/frameworks/acara-v9.json"
WORKBOOK = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = set("AC9E1LA06 AC9E1LA10 AC9E1LE03 AC9E1LY02 AC9E1LY04 AC9E1LY05 AC9E1LY06 AC9E1LY09 AC9E1LY11 AC9E1LY13".split())
CORE = {
    "a": "A", "is": "IS",  # explicitly taught familiar words in Weeks 1–2
    "bag": "b a g", "ben": "b e n", "bus": "b u s", "can": "c a n", "cap": "c a p",
    "cat": "c a t", "chat": "ch a t", "dog": "d o g", "dot": "d o t", "fish": "f i sh",
    "hen": "h e n", "hop": "h o p", "in": "i n", "lid": "l i d", "map": "m a p",
    "mat": "m a t", "meg": "m e g", "mend": "m e n d", "nap": "n a p", "net": "n e t",
    "on": "o n", "pat": "p a t", "pen": "p e n", "pot": "p o t", "red": "r e d",
    "rug": "r u g", "sam": "s a m", "ship": "sh i p", "shop": "sh o p", "sit": "s i t",
    "stand": "s t a n d", "step": "s t e p", "stop": "s t o p", "tap": "t a p",
    "tim": "t i m", "tin": "t i n", "tub": "t u b", "van": "v a n",
    # Fresh Route B check words, never in teaching materials before the check.
    "dash": "d a sh", "cash": "c a sh", "bush": "b u sh",
    "band": "b a n d", "tend": "t e n d", "land": "l a n d",
}
ROUTE_A = {
    "check": "ch e ck", "dock": "d o ck", "duck": "d u ck", "pack": "p a ck",
    "pick": "p i ck", "rock": "r o ck", "sack": "s a ck", "sock": "s o ck",
    # Fresh Route A check words.
    "tick": "t i ck", "lock": "l o ck", "kick": "k i ck",
    "rack": "r a ck", "muck": "m u ck", "puck": "p u ck",
}
HELD_OUT = set("dash cash bush band tend land tick lock kick rack muck puck".split())
DICTATION = set("bush kick land puck".split())


def require(test: bool, message: str) -> None:
    if not test:
        raise AssertionError(message)


def official_source() -> tuple[dict, dict[str, dict]]:
    require(IMPORT.is_file() and WORKBOOK.is_file(), "Pinned ACARA import/workbook missing")
    require(hashlib.sha256(WORKBOOK.read_bytes()).hexdigest() == SOURCE_SHA, "ACARA workbook SHA changed")
    data = json.loads(IMPORT.read_text(encoding="utf-8"))
    require(data["source_sha256"] == SOURCE_SHA, "ACARA import SHA changed")
    require(data["framework"] == "Australian Curriculum Version 9.0", "ACARA version changed")
    rows = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description"}
    for code in CODES:
        require(code in rows, f"Code absent from official workbook: {code}")
        attrs = rows[code]["attributes"]
        require(attrs["learning_area"] == "English" and attrs["level"] == "Year 1", f"Wrong level: {code}")
    return data, rows


def crosswalk(data: dict, rows: dict[str, dict]) -> str:
    evidence = {
        "AC9E1LA06": "Days 23–24, 28–29: one event/idea per sentence; partial",
        "AC9E1LA10": "Days 24, 28–30: familiar name capitals and full stops; punctuation range incomplete",
        "AC9E1LE03": "Days 21–22, 26–27: plot, character and setting of original teacher-read stories",
        "AC9E1LY02": "Days 21, 24, 27: paired question and response; not a full oral-language sample",
        "AC9E1LY04": "Days 21–30: route-gated controlled print, phrase and meaning; authentic texts elsewhere",
        "AC9E1LY05": "Days 21–22, 25–27, 30: heard-story evidence, retell and inference boundary",
        "AC9E1LY06": "Days 24–25, 28–30: short original sentence, reread and edit; spelling partial",
        "AC9E1LY09": "Day 21 and route checks: one-sound digraphs vs two-sound blends",
        "AC9E1LY11": "Days 21–30: taught short vowels, blends/digraphs and gated `ck`; long vowels/two syllables later",
        "AC9E1LY13": "Day 30: one-syllable dictated route word; broader spelling range later",
    }
    lines = [
        "# Australian Curriculum v9 · Year 1 English Weeks 5–6 crosswalk", "",
        "These are **exact Year 1 content-description links** to parts of ten lessons, not full description or achievement-standard coverage. Week 5 joins cumulative decodable print to original teacher-read literature; Week 6 retells plot/character/setting and makes short edited sentences. The school determines code order; authentic texts and broader Year 1 phonics/writing remain required.", "",
        f"Source: [official ACARA Version 9.0 workbook]({data['source_url']}), accessed {data['retrieved_at']}; workbook SHA-256 `{data['source_sha256']}`. Source rows below make the mapping independently checkable. Wording has whitespace normalised for plain-text display.", "",
        "| Official code | Workbook row | Level | Official content description | Pack evidence and boundary |",
        "|---|---:|---|---|---|",
    ]
    for code in sorted(CODES):
        row = rows[code]
        cells = [code, str(row["source_row"]), row["attributes"]["level"], " ".join(row["plain_text"].split()), evidence[code]]
        lines.append("| " + " | ".join(c.replace("|", "\\|") for c in cells) + " |")
    lines += [
        "",
        "© Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. Downloaded from the [Australian Curriculum website](https://www.australiancurriculum.edu.au/downloads) (accessed 29 September 2026) and modified for plain-text display. Curriculum material is licensed under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/); see [ACARA terms](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse SubjectNest; no affiliation, sponsorship or approval is claimed.", "",
        "The pinned import is a dated snapshot. Recheck current ACARA releases, state syllabus and local school sequence before implementation changes. See [source and review ledger](SOURCES-AND-REVIEW.md).", "",
    ]
    return "\n".join(lines)


def check_lessons() -> None:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    require(set(re.findall(r"\bAC9E1[A-Z0-9]+\b", content)) == CODES, "Lesson/crosswalk codes differ")
    days = [int(x) for x in re.findall(r"^## Day (\d+)\b", content, re.M)]
    require(days == list(range(21, 31)), f"Expected Days 21–30, got {days}")
    for day, section in zip(days, re.split(r"^## Day \d+\b", content, flags=re.M)[1:]):
        phases = re.findall(r"^([1-5])\. \*\*[^*]+ · (\d+) min\.\*\*", section, re.M)
        require(phases == [("1", "3"), ("2", "5"), ("3", "7"), ("4", "7"), ("5", "3")], f"Day {day} timing")
        require("**Access/home/extension:**" in section, f"Day {day} access missing")
    require("school's explicit teaching" in content and "Route A" in content, "School-sequence gate missing")


def word_check(word: str, route: str, location: str) -> None:
    word = word.lower()
    inventory = {**CORE, **ROUTE_A}
    require(word in inventory, f"Unreviewed child-print word `{word}` in {location}")
    parts = inventory[word].split()
    if word in {"a", "is"}:
        return
    require("".join(parts) == word, f"Bad grapheme map: {word}")
    require(sum(p in "aeiou" for p in parts) == 1, f"Expected one short vowel: {word}")
    require(all(len(p) == 1 or p in {"sh", "ch", "ck"} for p in parts), f"Unreviewed grapheme: {word}")
    if "ck" in parts:
        require(route == "A", f"`ck` leaked into core/Route B: {word} in {location}")
        require(parts[-1] == "ck" and parts[-2] in "aeiou", f"`ck` pattern mismatch: {word}")


def lines(path: Path) -> list[tuple[int, str, str]]:
    day, route = 0, "core"
    found = []
    for line in path.read_text(encoding="utf-8").splitlines():
        d = re.match(r"^## Day (\d+)\b", line)
        if d:
            day, route = int(d.group(1)), "core"
        if line.startswith("**Route A"):
            route = "A"
        if line.startswith("**Route B"):
            route = "B"
        match = re.search(r"\*\*Child reads:\*\* `([^`]+)`", line)
        if match:
            found.append((day, route, match.group(1)))
    return found


def check_print() -> None:
    practice = lines(ROOT / "learner-practice.md")
    checks = lines(ROOT / "student-checks.md")
    require(len(practice) == 40 and len({p[2] for p in practice}) == 40, "Expected 40 distinct daily child-print lines")
    for day in range(21, 31):
        require(sum(d == day and r == "core" for d, r, _ in practice) == 3, f"Day {day}: three core choices required")
        require(sum(d == day and r == "A" for d, r, _ in practice) == 1, f"Day {day}: one gated choice required")
    for day, route, sentence in practice + checks:
        require(21 <= day <= 30 and route in {"core", "A", "B"}, "Line lacks correct day/route")
        for word in re.findall(r"[A-Za-z]+", sentence):
            word_check(word, route, f"Day {day}")
    require({(d, r) for d, r, _ in checks} == {(25, "A"), (25, "B"), (30, "A"), (30, "B")}, "Four route-specific check lines required")
    checks_text = (ROOT / "student-checks.md").read_text(encoding="utf-8")
    require(not any(re.search(rf"\b{w}\b", checks_text, re.I) for w in DICTATION), "Dictated answer visible to learners")
    for day, route, sentence in checks:
        for word in re.findall(r"[A-Za-z]+", sentence):
            if word.lower() in HELD_OUT:
                require((day == 25 and word.lower() in {"dash", "tick"}) or (day == 30 and word.lower() in {"band", "rack"}), "Held-out item appears in wrong check")
    prior = "\n".join(p.read_text(encoding="utf-8") for folder in (ROOT.parent / "weeks-01-02", ROOT.parent / "weeks-03-04") for p in folder.rglob("*.md"))
    teaching = "\n".join((ROOT / p).read_text(encoding="utf-8") for p in ("README.md", "LESSONS.md", "learner-practice.md", "adult-read-stories.md", "print/TEXT-ALTERNATIVE.md"))
    for word in HELD_OUT:
        require(not re.search(rf"\b{word}\b", prior, re.I), f"Held-out `{word}` exposed in earlier pack")
        require(not re.search(rf"\b{word}\b", teaching, re.I), f"Held-out `{word}` exposed in teaching materials")
    require("teacher-key.md" not in checks_text and "teacher-key.md" not in (ROOT / "learner-practice.md").read_text(encoding="utf-8"), "Learner key link")


def check_links() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("http://", "https://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            require(target.exists(), f"Broken link {path.relative_to(ROOT)} -> {url}")
            count += 1
    return count


def check_aid() -> None:
    svg = ROOT / "print/story-and-print.svg"
    pdf = ROOT / "print/story-and-print.pdf"
    require(svg.is_file() and pdf.is_file(), "A4 aid missing")
    tree = ET.parse(svg)
    root = tree.getroot()
    require(root.attrib.get("width") == "210mm" and root.attrib.get("height") == "297mm", "SVG not A4")
    ns = {"s": "http://www.w3.org/2000/svg"}
    require(root.find("s:title", ns) is not None and root.find("s:desc", ns) is not None, "SVG needs accessible title/description")
    require("(A4)" in subprocess.check_output(["pdfinfo", str(pdf)], text=True), "PDF not A4")
    pdf_text = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
    require(all(s in pdf_text for s in ("STORY AND PRINT", "CHARACTER", "SETTING", "CHILD PRINT", "CC BY 4.0")), "PDF text missing")
    alternative = (ROOT / "print/TEXT-ALTERNATIVE.md").read_text(encoding="utf-8").lower()
    require(all(x in alternative for x in ("tactile", "five", "fact", "idea", "start", "end", "grapheme")), "Aid alternative incomplete")


def hashes() -> str:
    paths = sorted(p for p in ROOT.rglob("*") if p.is_file() and p.name != "MANIFEST.sha256" and "__pycache__" not in p.parts)
    return "".join(f"{hashlib.sha256(p.read_bytes()).hexdigest()}  {p.relative_to(ROOT)}\n" for p in paths)


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    data, rows = official_source()
    expected = crosswalk(data, rows)
    path = ROOT / "CURRICULUM-CROSSWALK.md"
    if args.write_crosswalk:
        path.write_text(expected, encoding="utf-8")
    require(path.read_text(encoding="utf-8") == expected, "Crosswalk differs from pinned official descriptions")
    manifest = ROOT / "MANIFEST.sha256"
    expected_hashes = hashes()
    if args.write_manifest:
        manifest.write_text(expected_hashes, encoding="utf-8")
    check_lessons()
    check_print()
    links = check_links()
    check_aid()
    require(manifest.read_text(encoding="utf-8") == expected_hashes, "Manifest stale or missing")
    print(f"PASS: 10 lessons/250 minutes, 40 distinct practice lines, 4 route-specific check lines, {len(CODES)} pinned official Year 1 English codes, {links} local links, A4 SVG/PDF/text alternative, held-out boundary, manifest")


if __name__ == "__main__":
    main()
