#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Recheck Year 8 English W03–04 content, source rows, aids and receipt."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import tempfile
import xml.etree.ElementTree as ET
from pathlib import Path

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
FRAMEWORK = STUDIO / "data/frameworks/acara-v9.json"
WORKBOOK = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
MANIFEST = ROOT / "MANIFEST.sha256"
ROWS = {
    "AC9E8LA02": 934, "AC9E8LA03": 937, "AC9E8LA04": 941,
    "AC9E8LA05": 944, "AC9E8LA06": 947, "AC9E8LA08": 952,
    "AC9E8LA09": 955, "AC9E8LE02": 964, "AC9E8LE03": 967,
    "AC9E8LE05": 973, "AC9E8LE06": 978, "AC9E8LY02": 989,
    "AC9E8LY03": 994, "AC9E8LY04": 997, "AC9E8LY05": 1002,
    "AC9E8LY06": 1007, "AC9E8LY07": 1012,
}
ASSETS = ("equivalence-evidence", "hybrid-source-map", "line-meaning")
TEACHING = ("README.md", "TEXTS.md", "LESSONS.md", "LEARNER-COPY.md", "OPTIONAL-PRACTICE-AND-HOME.md")


def need(condition: bool, message: str) -> None:
    if not condition:
        raise AssertionError(message)


def sha(path: Path) -> str:
    return hashlib.sha256(path.read_bytes()).hexdigest()


def official() -> dict[str, dict]:
    data = json.loads(FRAMEWORK.read_text(encoding="utf-8"))
    need(data["framework"] == "Australian Curriculum Version 9.0", "framework label drift")
    need(data["source_sha256"] == SOURCE_SHA == sha(WORKBOOK), "source workbook hash mismatch")
    selected = [r for r in data["records"] if r.get("code") in ROWS]
    need(len(selected) == len(ROWS) == 17, "seventeen unique official content-description rows required")
    by_code = {r["code"]: r for r in selected}
    for code, expected_row in ROWS.items():
        row = by_code[code]
        attr = row["attributes"]
        need(row["record_type"] == "content_description" and row["source_row"] == expected_row,
             f"{code}: source row/type mismatch")
        need(attr["learning_area"] == "English" and attr["level"] == "Year 8", f"{code}: area/level mismatch")
        need(" ".join(row["plain_text"].split()) == " ".join(attr["content_description"].split()),
             f"{code}: official wording drift")
    return by_code


def lesson_sections() -> dict[int, tuple[str, str]]:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) Day (\d+) (.+)$", content, re.MULTILINE))
    need([int(m.group(2)) for m in marks] == list(range(11, 21)), "lesson headers must be Days 11–20 once")
    sections = {}
    used = set()
    for i, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (3 if day <= 15 else 4), f"Day {day}: week mismatch")
        body = content[mark.end(): marks[i + 1].start() if i + 1 < len(marks) else len(content)]
        steps = [int(x) for x in re.findall(r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*", body, re.MULTILINE)]
        need(len(steps) == 6 and sum(steps) == 25, f"Day {day}: six timed steps must total 25, got {steps}")
        need("**Goal:**" in body and "**Prepare:**" in body and
             ("**Exit" in body or (day in (15, 20) and "**Collect" in body)),
             f"Day {day}: goal/preparation/exit absent")
        codes = set(re.findall(r"\bAC9E8[A-Z0-9]+\b", body))
        need(codes and codes <= ROWS.keys(), f"Day {day}: invalid/no code {codes - ROWS.keys()}")
        used |= codes
        sections[day] = (mark.group(3), body)
    need(used == ROWS.keys(), f"lesson codes differ from crosswalk: {used ^ ROWS.keys()}")
    need("listening-supported comprehension" in content and "not independent print decoding" in content,
         "decoding/access boundary missing")
    return sections


def crosswalk(by_code: dict[str, dict], sections: dict[int, tuple[str, str]]) -> str:
    lines = [
        "# ACARA v9 crosswalk · Year 8 English Weeks 3–4", "",
        "**Pinned source:** [Australian Curriculum Version 9.0 official workbook](https://www.australiancurriculum.edu.au/downloads), retrieved 29 September 2026, original XLSX SHA-256 `" + SOURCE_SHA + "`; exact source-row import in local `products/curriculum-studio/data/frameworks/acara-v9.json`. Wording below is the official content description after plain-text whitespace normalisation. All entries are **Year 8 English**. These are selected lesson opportunities, not whole-description completion, a national achievement decision or local syllabus approval.", "",
        "| Code | Original workbook row | Exact official Year 8 content description | Day(s) with a planned action |",
        "| --- | ---: | --- | --- |",
    ]
    for code, source_row in sorted(ROWS.items(), key=lambda entry: entry[1]):
        row = by_code[code]
        wording = " ".join(row["plain_text"].split()).replace("|", "\\|")
        days = ", ".join(f"{day} ({title})" for day, (title, body) in sections.items() if code in body)
        need(days != "", f"{code}: no lesson opportunity")
        lines.append(f"| `{code}` | {source_row} | {wording} | {days} |")
    lines += [
        "", "## How to interpret these links", "",
        "Codes identify a **part** of a description addressed by a task. Source evaluation uses fictional paper display and timing models; no real site, event, wall or visitor has been studied. The literary work includes one original vignette and poem; the simile/metaphor, viewpoint and punctuation tasks sample only a small part of the language/literature descriptions. A school must plan a far wider range of Australian, First Nations Australian and world-authored literature with rights and cultural protocols; original SubjectNest texts cannot replace this. Creation codes receive draft, edit and private reader-test opportunities, not the full range of purposes, modes or publication. LY07 has actual private spoken delivery only for a learner who takes that route; an AAC or written response is recorded as that route rather than invented vocal evidence. A read-aloud response is listening-supported comprehension, not independently decoded reading.", "",
        "The [Day 15 and 20 checks](ASSESSMENT.md) sample transfer to new texts. The separate [staff key](TEACHER-KEY.md) gives criterion-level next moves; neither check alone decides the achievement standard. State/territory scope, school timetable, accessibility and text selection require local review.", "",
        "**ACARA attribution:** © Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. This material was downloaded from the [Australian Curriculum website](https://www.australiancurriculum.edu.au/downloads) (accessed 29 September 2026) and modified only for plain-text whitespace and selected rows, under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/) subject to [ACARA terms and exclusions](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse SubjectNest. Original crosswalk commentary © NeuroForgeIO Pty Ltd 2026, CC BY 4.0.", "",
    ]
    return "\n".join(lines)


def learner_and_checks() -> None:
    learner = (ROOT / "LEARNER-COPY.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\s*$", learner, re.MULTILINE))
    need([int(m.group(1)) for m in marks] == list(range(11, 21)), "learner Days 11–20")
    for i, mark in enumerate(marks):
        day = int(mark.group(1))
        body = learner[mark.end(): marks[i + 1].start() if i + 1 < len(marks) else len(learner)]
        need([x for x in re.findall(r"^- \*\*([ABC]) ·", body, re.MULTILINE)] == ["A", "B", "C"],
             f"Day {day}: three A/B/C choices")
        need("**Exit:**" in body, f"Day {day}: missing exit")
    need("TEACHER-KEY" not in learner and "ASSESSMENT.md" not in learner,
         "child-facing page links private check/key")
    practice = (ROOT / "OPTIONAL-PRACTICE-AND-HOME.md").read_text(encoding="utf-8")
    rows = re.findall(r"^\| (\d{2}) \| (.+) \| (.+) \| (.+) \|$", practice, re.MULTILINE)
    need([int(row[0]) for row in rows] == list(range(11, 21)) and
         all(all(cell.strip() for cell in row[1:]) for row in rows),
         "ten daily optional practice/home rows")
    need(practice.count("five minutes") >= 2 and "No paid account, device" in practice,
         "extras/home boundary missing")
    teaching_text = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in TEACHING)
    for held_out in ("20 + 16c", "four-card result proves", "7 + 11b", "paper copy for everyone", "distribution will be arranged"):
        need(held_out.lower() not in teaching_text.lower(), f"held-out phrase leaked into teaching: {held_out}")
    checks = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    need(all(x in checks for x in ("Check A Day 15", "Check B Day 20", "20 + 16c", "7 + 11b")),
         "fresh assessment source incomplete")
    for code in ("A1", "A2", "A3", "A4", "B1", "B2", "B3", "B4"):
        need(re.search(rf"^\| {code} ·", key, re.MULTILINE) is not None, f"rubric {code} absent")
    need("listening comprehension" in checks and "not independent print-reading evidence" in checks,
         "assessment access boundary missing")
    need("**A · Spoken presentation.**" in learner and
         "**B · AAC/text-to-speech presentation.**" in learner and
         "**C · Written alternative.**" in learner and
         "not a spoken performance sample" in learner,
         "delivery evidence must follow actual Day 19 mode")
    need(12 + 18 * 2 + 12 == 60 and 24 + 18 * 4 == 96 and 24 + 18 * 5 == 114 and
         5 + 15 * 3 == 50 and 5 + 15 * 4 == 65 and
         10 + 16 * 4 + 10 == 84 and 20 + 16 * 5 == 100 and
         7 + 11 * 3 == 40 and 7 + 11 * 4 == 51 and
         3 + 2 * 4 + 3 == 6 + 2 * 4 and 4 + 10 * 2 + 4 == 8 + 10 * 2 and
         5 + 10 * 4 == 45 and 5 + 10 * 5 > 45,
         "arithmetic invariant failed")


def assets() -> None:
    alt = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    ns = {"s": "http://www.w3.org/2000/svg"}
    heads = ("## Equivalence and evidence aid", "## Hybrid source map", "## Line and meaning aid")
    need(all(h in alt for h in heads), "three full text/tactile sections required")
    for stem in ASSETS:
        svg = ROOT / "print" / f"{stem}.svg"
        pdf = svg.with_suffix(".pdf")
        need(svg.exists() and pdf.exists(), f"{stem}: missing SVG/PDF")
        root = ET.parse(svg).getroot()
        need(root.attrib.get("width") == "210mm" and root.attrib.get("height") == "297mm" and
             root.attrib.get("viewBox") == "0 0 210 297" and root.attrib.get("role") == "img" and
             root.attrib.get("aria-labelledby") == "title desc", f"{stem}: A4/access metadata")
        need(root.find("s:title", ns) is not None and root.find("s:desc", ns) is not None, f"{stem}: title/desc")
        need("CC BY 4.0" in svg.read_text(encoding="utf-8"), f"{stem}: SVG rights")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        need("(A4)" in info and re.search(r"Pages:\s+1\b", info) is not None, f"{stem}: PDF geometry")
        extracted = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        need("SubjectNest" in extracted and "CC BY 4.0" in extracted and len(extracted) > 160,
             f"{stem}: PDF text extraction")
        fonts = subprocess.check_output(["pdffonts", str(pdf)], text=True)
        need("DejaVuSans" in fonts, f"{stem}: expected embedded DejaVu subset")
    with tempfile.TemporaryDirectory(prefix="subjectnest-y8-english-") as temporary:
        subprocess.run([sys.executable, str(ROOT / "print/generate_print.py"), "--output-dir", temporary],
                       check=True, capture_output=True, text=True)
        for stem in ASSETS:
            for suffix in (".svg", ".pdf"):
                name = stem + suffix
                need(sha(Path(temporary) / name) == sha(ROOT / "print" / name),
                     f"{name}: rendered bytes differ from original source")


def slug(heading: str) -> str:
    heading = re.sub(r"<[^>]+>", "", heading)
    return re.sub(r"[^a-z0-9-]", "", heading.strip().lower().replace(" ", "-"))


def links_and_rights() -> int:
    checked = 0
    for path in ROOT.rglob("*.md"):
        source = path.read_text(encoding="utf-8")
        need("CC BY 4.0" in source, f"{path}: no original/curriculum licence note")
        for raw in re.findall(r"\[[^]]+\]\(([^)]+)\)", source):
            if raw.startswith(("https://", "http://")):
                continue
            relative, _, anchor = raw.partition("#")
            target = (path.parent / relative).resolve() if relative else path
            need(target.exists(), f"{path}: broken link {raw}")
            if anchor and target.suffix.lower() == ".md":
                headings = [slug(m.group(1)) for m in re.finditer(r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)]
                need(anchor in headings, f"{path}: broken heading #{anchor} in {target}")
            checked += 1
    return checked


def payload_files() -> list[Path]:
    return sorted((p for p in ROOT.rglob("*") if p.is_file() and p != MANIFEST and
                   "__pycache__" not in p.parts and ".ruff_cache" not in p.parts),
                  key=lambda p: p.relative_to(ROOT).as_posix())


def manifest() -> str:
    return "".join(f"{sha(p)}  {p.relative_to(ROOT).as_posix()}\n" for p in payload_files())


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    by_code = official()
    sections = lesson_sections()
    expected = crosswalk(by_code, sections)
    path = ROOT / "CURRICULUM-CROSSWALK.md"
    if args.write_crosswalk:
        path.write_text(expected, encoding="utf-8")
    else:
        need(path.read_text(encoding="utf-8") == expected, "crosswalk official wording/day drift")
    learner_and_checks()
    assets()
    link_count = links_and_rights()
    expected_manifest = manifest()
    if args.write_manifest:
        MANIFEST.write_text(expected_manifest, encoding="utf-8")
    else:
        need(MANIFEST.read_text(encoding="utf-8") == expected_manifest, "SHA manifest drift")
    print(f"PASS: 10 × 25-minute lessons; 30 core routes; 20 optional extras + 10 home routes; "
          f"2 fresh checks/8 rubric criteria; 17 exact Year 8 English codes/rows; "
          f"3 A4 SVG/PDF/text aids; {link_count} local links; {len(payload_files())} hashed files")


if __name__ == "__main__":
    main()
