#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only checks for Year 1 English Weeks 3–4; flags refresh derived files."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
IMPORT = ROOT.parents[4] / "data/frameworks/acara-v9.json"
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = set("AC9E1LA06 AC9E1LA10 AC9E1LE03 AC9E1LY02 AC9E1LY04 AC9E1LY05 AC9E1LY06 AC9E1LY09 AC9E1LY11 AC9E1LY13".split())
GRAMMAR = {
    # The only exceptional high-frequency print here was explicitly taught in Weeks 1–2.
    "a": ("EXPLICIT_A",), "is": ("EXPLICIT_IS",),
    "sam": ("s", "a", "m"), "tim": ("t", "i", "m"),
    "chad": ("ch", "a", "d"), "stan": ("s", "t", "a", "n"),
    "bag": ("b", "a", "g"), "can": ("c", "a", "n"),
    "cat": ("c", "a", "t"), "chat": ("ch", "a", "t"),
    "chop": ("ch", "o", "p"), "dog": ("d", "o", "g"),
    "fish": ("f", "i", "sh"), "hen": ("h", "e", "n"),
    "in": ("i", "n"), "map": ("m", "a", "p"),
    "mat": ("m", "a", "t"), "mend": ("m", "e", "n", "d"),
    "net": ("n", "e", "t"), "on": ("o", "n"),
    "pen": ("p", "e", "n"), "ship": ("sh", "i", "p"),
    "shop": ("sh", "o", "p"), "sit": ("s", "i", "t"),
    "stand": ("s", "t", "a", "n", "d"), "step": ("s", "t", "e", "p"),
    "stop": ("s", "t", "o", "p"), "tap": ("t", "a", "p"),
    "tin": ("t", "i", "n"), "bend": ("b", "e", "n", "d"),
    "wish": ("w", "i", "sh"), "chum": ("ch", "u", "m"),
    "mash": ("m", "a", "sh"), "mush": ("m", "u", "sh"),
    "much": ("m", "u", "ch"), "rush": ("r", "u", "sh"),
    # Route A only, after Day 16's explicit /k/ <ck> model.
    "check": ("ch", "e", "ck"), "dock": ("d", "o", "ck"),
    "duck": ("d", "u", "ck"), "luck": ("l", "u", "ck"),
    "pack": ("p", "a", "ck"), "peck": ("p", "e", "ck"),
    "pick": ("p", "i", "ck"), "rock": ("r", "o", "ck"),
    "sack": ("s", "a", "ck"), "sock": ("s", "o", "ck"),
    "tuck": ("t", "u", "ck"),
}
BEFORE_CHECK_15 = {"wish", "chum", "mash"}
BEFORE_CHECK_20 = {"luck", "peck", "tuck", "mush", "much", "rush"}
PRIOR = ROOT.parent / "weeks-01-02"


def require(condition: bool, message: str) -> None:
    if not condition:
        raise AssertionError(message)


def source() -> tuple[dict, dict[str, dict]]:
    require(IMPORT.is_file(), f"Missing pinned ACARA import: {IMPORT}")
    data = json.loads(IMPORT.read_text(encoding="utf-8"))
    require(data["source_sha256"] == SOURCE_SHA, "Pinned workbook SHA changed")
    require(data["framework"] == "Australian Curriculum Version 9.0", "Framework version changed")
    require("ac-version-9" in data["source_url"], "Workbook URL changed")
    descriptions = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description"}
    for code in CODES:
        require(code in descriptions, f"Unknown code: {code}")
        r = descriptions[code]
        require(r["attributes"]["learning_area"] == "English" and r["attributes"]["level"] == "Year 1", f"Wrong level/area: {code}")
    return data, descriptions


def norm(value: str) -> str:
    return " ".join(value.split())


def expected_crosswalk(data: dict, descriptions: dict[str, dict]) -> str:
    lines = [
        "# Australian Curriculum v9 · Year 1 English Weeks 3–4 crosswalk",
        "",
        "Each code is an **exact Year 1 content-description link** to the lessons, not a claim of full content or achievement-standard coverage. Week 3 consolidates taught graphemes and oral retell. Week 4 gates `ck` by the school's actual scope and teaches sentence boundaries. Authentic literature, later phonics and sustained writing are still required.",
        "",
        f"Source: [official ACARA Version 9.0 workbook]({data['source_url']}), accessed {data['retrieved_at']}; workbook SHA-256 `{data['source_sha256']}`. Source rows below allow rechecking. Wording is preserved with whitespace normalised for plain-text display.",
        "",
        "| Official code | Workbook row | Level | Official content description | Pack evidence and boundary |",
        "|---|---:|---|---|---|",
    ]
    evidence = {
        "AC9E1LA06": "Days 12, 14, 18: one complete actor/action idea; partial",
        "AC9E1LA10": "Days 18, 20: familiar names, full stops; punctuation range incomplete",
        "AC9E1LE03": "Days 13, 14, 19: adult-read plot, character and setting talk",
        "AC9E1LY02": "Day 13: partner turn/response; not a full oral-language sample",
        "AC9E1LY04": "Days 11–20: controlled print and reread; authentic texts remain elsewhere",
        "AC9E1LY05": "Days 13–15, 19–20: adult-read retell and text questions",
        "AC9E1LY06": "Days 14, 18–20: short idea, reread and boundary edit; spelling partial",
        "AC9E1LY09": "Days 12, 16: phoneme boxes for blends/digraphs",
        "AC9E1LY11": "Days 11–12, 15–17, 20: taught short-vowel code and route; long vowels/two syllables later",
        "AC9E1LY13": "Day 17: route word spelling; one- and two-syllable range incomplete",
    }
    for code in sorted(CODES):
        r = descriptions[code]
        cells = [code, str(r["source_row"]), r["attributes"]["level"], norm(r["plain_text"]), evidence[code]]
        lines.append("| " + " | ".join(c.replace("|", "\\|") for c in cells) + " |")
    lines += [
        "",
        "© Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. Downloaded from the Australian Curriculum website (accessed 29 September 2026) and modified for plain-text display. Curriculum material is licensed under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/); see [ACARA terms](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse SubjectNest; no affiliation, sponsorship or approval is claimed.",
        "",
        "The pinned import is a dated snapshot; refresh against current ACARA releases before later publication changes. See [source and review ledger](SOURCES-AND-REVIEW.md).",
        "",
    ]
    return "\n".join(lines)


def check_scripts() -> None:
    lessons = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    require(set(re.findall(r"\bAC9E1[A-Z0-9]+\b", lessons)) == CODES, "Lesson code set differs from official crosswalk")
    days = [int(x) for x in re.findall(r"^## Day (\d+)\b", lessons, re.M)]
    require(days == list(range(11, 21)), "Ten scripts must be Days 11–20")
    sections = re.split(r"^## Day \d+\b", lessons, flags=re.M)[1:]
    for day, section in enumerate(sections, 11):
        phases = re.findall(r"^([1-5])\. \*\*[^*]+ · (\d+) min\.\*\*", section, re.M)
        require(phases == [("1", "3"), ("2", "5"), ("3", "7"), ("4", "7"), ("5", "3")], f"Day {day}: timing must total 25")
        require("**Access/home/extension:**" in section, f"Day {day}: missing access/home/extension")
    require("Route A" in lessons and "Route B" in lessons and "school's phonics scope" in lessons, "School-sequence gate missing")


def word_parts(word: str, day: int, route: str, path: str) -> None:
    w = word.lower()
    require(w in GRAMMAR, f"Untaught/unreviewed word `{word}` in {path} Day {day} {route}")
    parts = GRAMMAR[w]
    if w not in ("a", "is"):
        require("".join(parts) == w, f"Bad sound-box ledger: {word}")
        require(sum(p in "aeiou" for p in parts) == 1, f"Expected one short vowel: {word}")
        require(all(len(p) == 1 or p in {"sh", "ch", "ck"} for p in parts), f"Unreviewed digraph: {word}")
    if "ck" in parts:
        require(day >= 16 and route == "A", f"`ck` before teaching or in Route B: {word} Day {day} {route}")
        i = parts.index("ck")
        require(i == len(parts) - 1 and parts[i - 1] in "aeiou", f"`ck` pattern mismatch: {word}")


def controlled_lines(path: Path) -> list[tuple[int, str, str]]:
    day = 0
    route = "core"
    lines: list[tuple[int, str, str]] = []
    for line in path.read_text(encoding="utf-8").splitlines():
        match = re.match(r"^## Day (\d+)\b", line)
        if match:
            day = int(match.group(1))
            route = "core"
        if re.match(r"^(?:\*\*|### )Route A", line):
            route = "A"
        elif re.match(r"^(?:\*\*|### )Route B", line):
            route = "B"
        for label in ("Child reads", "Repair this printed message"):
            match = re.search(rf"\*\*{label}:\*\* `([^`]+)`", line)
            if match:
                require(day in range(11, 21), f"Missing day for controlled line in {path.name}")
                lines.append((day, route, match.group(1)))
    return lines


def check_decodability() -> tuple[int, int]:
    practice = ROOT / "learner-practice.md"
    checks = ROOT / "student-checks.md"
    practice_lines = controlled_lines(practice)
    check_lines = controlled_lines(checks)
    require(len(practice_lines) == 45, f"Expected 45 three-choice practice lines, got {len(practice_lines)}")
    counts = {(d, r): 0 for d in range(11, 21) for r in ("core", "A", "B")}
    for day, route, sentence in practice_lines:
        counts[(day, route)] += 1
        for word in re.findall(r"[A-Za-z]+", sentence):
            word_parts(word, day, route, practice.name)
    for day in range(11, 16):
        require(counts[(day, "core")] == 3, f"Day {day}: expected 3 contexts")
    for day in range(16, 21):
        require(counts[(day, "A")] == 3 and counts[(day, "B")] == 3, f"Day {day}: expected 3 per route")
    require(len(check_lines) == 11, f"Expected 11 controlled check/read/edit lines, got {len(check_lines)}")
    for day, route, sentence in check_lines:
        for word in re.findall(r"[A-Za-z]+", sentence):
            word_parts(word, day, route, checks.name)
    for word in ("mash", "tuck", "rush"):
        require(word in GRAMMAR, f"Dictation ledger missing {word}")
    # Held-out items may be named in the held-out check/teacher key, never in earlier teaching materials.
    prior_material = "\n".join(p.read_text(encoding="utf-8") for p in PRIOR.glob("*.md"))
    before = "\n".join((ROOT / p).read_text(encoding="utf-8") for p in ("LESSONS.md", "learner-practice.md", "adult-read-stories.md", "print/TEXT-ALTERNATIVE.md"))
    for word in BEFORE_CHECK_15 | BEFORE_CHECK_20:
        require(not re.search(rf"\b{word}\b", prior_material, re.I), f"Held-out `{word}` appeared in Weeks 1–2")
        require(not re.search(rf"\b{word}\b", before, re.I), f"Held-out `{word}` leaked into lesson/practice/story")
    for p in (practice, checks):
        content = p.read_text(encoding="utf-8")
        require("teacher-key.md" not in content and "answer key" not in content.lower(), f"Learner link/key leakage: {p.name}")
    return len(practice_lines), len(check_lines)


def check_links() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("http://", "https://", "mailto:", "#")):
                continue
            target, _, fragment = url.partition("#")
            local = (path.parent / unquote(target)).resolve()
            require(local.is_file(), f"Broken local link: {path.relative_to(ROOT)} -> {url}")
            count += 1
    return count


def check_aid() -> None:
    svg = ROOT / "print/sound-to-sentence.svg"
    pdf = ROOT / "print/sound-to-sentence.pdf"
    require(svg.is_file() and pdf.is_file(), "Missing aid")
    tree = ET.parse(svg)
    root = tree.getroot()
    ns = {"s": "http://www.w3.org/2000/svg"}
    require(root.attrib["width"] == "210mm" and root.attrib["height"] == "297mm", "Aid SVG is not A4")
    require(root.find("s:title", ns) is not None and root.find("s:desc", ns) is not None, "Aid SVG access metadata")
    require("SOUND 5" in svg.read_text(encoding="utf-8"), "Aid needs five boxes for `stand`")
    info = subprocess.run(["pdfinfo", str(pdf)], capture_output=True, text=True, check=True).stdout
    require("(A4)" in info, "Aid PDF is not A4")
    searchable = subprocess.run(["pdftotext", str(pdf), "-"], capture_output=True, text=True, check=True).stdout
    require("From sounds to a sentence" in searchable and "CC BY 4.0" in searchable, "PDF text or credit missing")
    fonts = subprocess.run(["pdffonts", str(pdf)], capture_output=True, text=True, check=True).stdout
    require("DejaVu" in fonts, "Embedded DejaVu subset missing")
    alternative = (ROOT / "print/TEXT-ALTERNATIVE.md").read_text(encoding="utf-8")
    require(all(k in alternative.casefold() for k in ("say", "map", "blend", "box 5", "first", "next", "last", "sentence", "tactile")), "Text alternative incomplete")


def hashes() -> str:
    files = sorted(p for p in ROOT.rglob("*") if p.is_file() and p.name != "MANIFEST.sha256" and not p.name.startswith("_preview") and "__pycache__" not in p.parts)
    return "".join(f"{hashlib.sha256(p.read_bytes()).hexdigest()}  {p.relative_to(ROOT)}\n" for p in files)


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    data, descriptions = source()
    crosswalk = ROOT / "CURRICULUM-CROSSWALK.md"
    expected = expected_crosswalk(data, descriptions)
    if args.write_crosswalk:
        crosswalk.write_text(expected, encoding="utf-8")
    require(crosswalk.read_text(encoding="utf-8") == expected, "Official crosswalk drift")
    manifest = ROOT / "MANIFEST.sha256"
    if args.write_manifest:
        manifest.touch(exist_ok=True)
    check_scripts()
    pcount, ccount = check_decodability()
    links = check_links()
    check_aid()
    expected_hashes = hashes()
    if args.write_manifest:
        manifest.write_text(expected_hashes, encoding="utf-8")
    require(manifest.read_text(encoding="utf-8") == expected_hashes, "Hash manifest drift")
    print(f"PASS: 10 × 25-minute scripts · {pcount} controlled practice lines · {ccount} controlled check/edit lines · {len(CODES)} official codes · A4 aid · {links} local links · {len(expected_hashes.splitlines())} hashes")


if __name__ == "__main__":
    try:
        main()
    except (AssertionError, KeyError, FileNotFoundError, subprocess.CalledProcessError) as exc:
        print(f"FAIL: {exc}", file=sys.stderr)
        raise SystemExit(1)
