#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Recheck Year 7 English W03–04 content, source rows, aids and receipt."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import tempfile
import xml.etree.ElementTree as ET
from fractions import Fraction
from pathlib import Path

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
FRAMEWORK = STUDIO / "data/frameworks/acara-v9.json"
WORKBOOK = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
MANIFEST = ROOT / "MANIFEST.sha256"
ROWS = {
    "AC9E7LA02": 831, "AC9E7LA03": 836, "AC9E7LA04": 840,
    "AC9E7LA05": 844, "AC9E7LA06": 848, "AC9E7LA08": 855,
    "AC9E7LA09": 858, "AC9E7LE03": 871, "AC9E7LE05": 878,
    "AC9E7LE06": 881, "AC9E7LY02": 899, "AC9E7LY03": 906,
    "AC9E7LY04": 909, "AC9E7LY05": 911, "AC9E7LY06": 916,
}
ASSETS = ("claim-reach-mat", "paragraph-path", "number-meaning-map")
TEACHING = ("README.md", "TEXTS.md", "LESSONS.md", "LEARNER-COPY.md", "OPTIONAL-PRACTICE-AND-HOME.md")


def need(condition: bool, message: str) -> None:
    if not condition:
        raise AssertionError(message)


def sha(path: Path) -> str:
    return hashlib.sha256(path.read_bytes()).hexdigest()


def official() -> dict[str, dict]:
    data = json.loads(FRAMEWORK.read_text(encoding="utf-8"))
    need(data["framework"] == "Australian Curriculum Version 9.0", "framework label drift")
    need(data["source_sha256"] == SOURCE_SHA == sha(WORKBOOK), "source workbook hash mismatch")
    selected = [r for r in data["records"] if r.get("code") in ROWS]
    need(len(selected) == len(ROWS) == 15, "fifteen unique official content-description rows required")
    by_code = {r["code"]: r for r in selected}
    for code, expected_row in ROWS.items():
        row = by_code[code]
        attr = row["attributes"]
        need(row["record_type"] == "content_description" and row["source_row"] == expected_row,
             f"{code}: source row/type mismatch")
        need(attr["learning_area"] == "English" and attr["level"] == "Year 7", f"{code}: area/level mismatch")
        need(" ".join(row["plain_text"].split()) == " ".join(attr["content_description"].split()),
             f"{code}: official wording drift")
    return by_code


def lesson_sections() -> dict[int, tuple[str, str]]:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+) — (.+)$", content, re.MULTILINE))
    need([int(m.group(2)) for m in marks] == list(range(11, 21)), "lesson headers must be Days 11–20 once")
    sections = {}
    used = set()
    for i, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (3 if day <= 15 else 4), f"Day {day}: week mismatch")
        body = content[mark.end(): marks[i + 1].start() if i + 1 < len(marks) else len(content)]
        steps = [int(x) for x in re.findall(r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*", body, re.MULTILINE)]
        need(len(steps) == 6 and sum(steps) == 25, f"Day {day}: six timed steps must total 25, got {steps}")
        need("**Goal:**" in body and "**Prepare:**" in body and
             ("**Exit" in body or (day in (15, 20) and "**Collect" in body)),
             f"Day {day}: goal/preparation/exit absent")
        codes = set(re.findall(r"\bAC9E7[A-Z0-9]+\b", body))
        need(codes and codes <= ROWS.keys(), f"Day {day}: invalid/no code {codes - ROWS.keys()}")
        used |= codes
        sections[day] = (mark.group(3), body)
    need(used == ROWS.keys(), f"lesson codes differ from crosswalk: {used ^ ROWS.keys()}")
    need("listening-supported comprehension" in content and "not proof of independent print reading" in content,
         "decoding/access boundary missing")
    return sections


def crosswalk(by_code: dict[str, dict], sections: dict[int, tuple[str, str]]) -> str:
    lines = [
        "# ACARA v9 crosswalk · Year 7 English Weeks 3–4", "",
        "**Pinned source:** [Australian Curriculum Version 9.0 official workbook](https://www.australiancurriculum.edu.au/downloads), retrieved 29 September 2026, original XLSX SHA-256 `" + SOURCE_SHA + "`; exact source-row import in the [local framework](../../../../../data/frameworks/acara-v9.json). Wording below is the official content description after plain-text whitespace normalisation. All entries are **Year 7 English**. These are selected lesson opportunities, not whole-description completion, a national achievement decision or local syllabus approval.", "",
        "| Code | Original workbook row | Exact official Year 7 content description | Day(s) with a planned action |",
        "| --- | ---: | --- | --- |",
    ]
    for code, source_row in sorted(ROWS.items(), key=lambda entry: entry[1]):
        row = by_code[code]
        wording = " ".join(row["plain_text"].split()).replace("|", "\\|")
        days = ", ".join(f"{day} ({title})" for day, (title, body) in sections.items() if code in body)
        need(days != "", f"{code}: no lesson opportunity")
        lines.append(f"| `{code}` | {source_row} | {wording} | {days} |")
    lines += [
        "", "## How to interpret these links", "",
        "Codes identify a **part** of a description addressed by a task. Source evaluation uses fictional paper logs and comments; it does not establish actual biology, transport or product claims. Language analysis reaches paired captions, the technical/everyday uses of number words, one original literary vignette and one poem; Year 7's much wider requirement for works by First Nations Australian and diverse Australian/world authors remains for a school to plan and source with rights and protocols. Creation codes receive draft, edit and private reader-test opportunities, not a blanket claim that all modes/purposes or publication requirements are complete. Spoken discussion is observed only where it actually occurs; an AAC or written route is captured as that route. A read-aloud response is listening-supported comprehension, not independently decoded reading.", "",
        "The [Day 15 and 20 checks](ASSESSMENT.md) sample transfer to new texts. The separate [staff key](TEACHER-KEY.md) gives criterion-level next moves; neither check alone decides the achievement standard. State/territory scope, school timetable, accessibility and text selection require local review.", "",
        "**ACARA attribution:** © Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. This material was downloaded from the [Australian Curriculum website](https://www.australiancurriculum.edu.au/downloads) (accessed 29 September 2026) and modified only for plain-text whitespace and selected rows, under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/) subject to [ACARA terms and exclusions](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse SubjectNest. Original crosswalk commentary © NeuroForgeIO Pty Ltd 2026, CC BY 4.0.", "",
    ]
    return "\n".join(lines)


def learner_and_checks() -> None:
    learner = (ROOT / "LEARNER-COPY.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\s*$", learner, re.MULTILINE))
    need([int(m.group(1)) for m in marks] == list(range(11, 21)), "learner Days 11–20")
    for i, mark in enumerate(marks):
        day = int(mark.group(1))
        body = learner[mark.end(): marks[i + 1].start() if i + 1 < len(marks) else len(learner)]
        need([x for x in re.findall(r"^- \*\*([ABC]) ·", body, re.MULTILINE)] == ["A", "B", "C"],
             f"Day {day}: three A/B/C choices")
        need("**Exit:**" in body, f"Day {day}: missing exit")
    need("TEACHER-KEY" not in learner and "ASSESSMENT.md" not in learner,
         "child-facing page links private check/key")
    practice = (ROOT / "OPTIONAL-PRACTICE-AND-HOME.md").read_text(encoding="utf-8")
    need([int(x) for x in re.findall(r"^\| (\d{2}) \|", practice, re.MULTILINE)] == list(range(11, 21)),
         "ten daily optional practice/home rows")
    need(practice.count("five minutes") >= 2 and "No device, internet" in practice,
         "extras/home boundary missing")
    teaching_text = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in TEACHING)
    for held_out in ("pretend parcel desk", "new **star** card", "16 of 40 equal paper cells", "diagram does the talking"):
        need(held_out.lower() not in teaching_text.lower(), f"held-out phrase leaked into teaching: {held_out}")
    checks = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    need(all(x in checks for x in ("Check A · Day 15", "Check B · Day 20", "new **star** card", "16 of 40 equal paper cells")),
         "fresh assessment source incomplete")
    for code in ("A1", "A2", "A3", "A4", "B1", "B2", "B3", "B4"):
        need(re.search(rf"^\| {code} ·", key, re.MULTILINE) is not None, f"rubric {code} absent")
    need("listening-supported comprehension" in checks and "not count it as evidence of independent print decoding" in checks,
         "assessment access boundary missing")
    need(Fraction(16, 40) == Fraction(2, 5) == Fraction(40, 100) and
         36 == 6 * 6 == 4 * 9 == 2**2 * 3**2 and
         Fraction(36, 100) * 50 == 18 and Fraction(3, 4) == Fraction(75, 100),
         "arithmetic invariant failed")


def assets() -> None:
    alt = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    ns = {"s": "http://www.w3.org/2000/svg"}
    heads = ("## Claim and source-reach mat", "## Paragraph-path aid", "## Number-and-meaning map")
    need(all(h in alt for h in heads), "three full text/tactile sections required")
    for stem in ASSETS:
        svg = ROOT / "print" / f"{stem}.svg"
        pdf = svg.with_suffix(".pdf")
        need(svg.exists() and pdf.exists(), f"{stem}: missing SVG/PDF")
        root = ET.parse(svg).getroot()
        need(root.attrib.get("width") == "210mm" and root.attrib.get("height") == "297mm" and
             root.attrib.get("viewBox") == "0 0 210 297" and root.attrib.get("role") == "img" and
             root.attrib.get("aria-labelledby") == "title desc", f"{stem}: A4/access metadata")
        need(root.find("s:title", ns) is not None and root.find("s:desc", ns) is not None, f"{stem}: title/desc")
        need("CC BY 4.0" in svg.read_text(encoding="utf-8"), f"{stem}: SVG rights")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        need("(A4)" in info and re.search(r"Pages:\s+1\b", info) is not None, f"{stem}: PDF geometry")
        extracted = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        need("SubjectNest" in extracted and "CC BY 4.0" in extracted and len(extracted) > 160,
             f"{stem}: PDF text extraction")
        fonts = subprocess.check_output(["pdffonts", str(pdf)], text=True)
        need("DejaVuSans" in fonts, f"{stem}: expected embedded DejaVu subset")
    with tempfile.TemporaryDirectory(prefix="subjectnest-y7-english-") as temporary:
        subprocess.run([sys.executable, str(ROOT / "print/generate_print.py"), "--output-dir", temporary],
                       check=True, capture_output=True, text=True)
        for stem in ASSETS:
            for suffix in (".svg", ".pdf"):
                name = stem + suffix
                need(sha(Path(temporary) / name) == sha(ROOT / "print" / name),
                     f"{name}: rendered bytes differ from original source")


def slug(heading: str) -> str:
    heading = re.sub(r"<[^>]+>", "", heading)
    # Match the published Markdown heading IDs: remove punctuation before
    # collapsing the surrounding whitespace to one hyphen.
    return re.sub(r"\s+", "-", re.sub(r"[^\w\- ]", "", heading.lower().replace("`", ""))).strip("-")


def links_and_rights() -> int:
    checked = 0
    for path in ROOT.rglob("*.md"):
        source = path.read_text(encoding="utf-8")
        need("CC BY 4.0" in source, f"{path}: no original/curriculum licence note")
        for raw in re.findall(r"\[[^]]+\]\(([^)]+)\)", source):
            if raw.startswith(("https://", "http://")):
                continue
            relative, _, anchor = raw.partition("#")
            target = (path.parent / relative).resolve() if relative else path
            need(target.exists(), f"{path}: broken link {raw}")
            if anchor and target.suffix.lower() == ".md":
                headings = [slug(m.group(1)) for m in re.finditer(r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)]
                need(anchor in headings, f"{path}: broken heading #{anchor} in {target}")
            checked += 1
    return checked


def payload_files() -> list[Path]:
    return sorted((p for p in ROOT.rglob("*") if p.is_file() and p != MANIFEST and
                   "__pycache__" not in p.parts and ".ruff_cache" not in p.parts),
                  key=lambda p: p.relative_to(ROOT).as_posix())


def manifest() -> str:
    return "".join(f"{sha(p)}  {p.relative_to(ROOT).as_posix()}\n" for p in payload_files())


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    by_code = official()
    sections = lesson_sections()
    expected = crosswalk(by_code, sections)
    path = ROOT / "CURRICULUM-CROSSWALK.md"
    if args.write_crosswalk:
        path.write_text(expected, encoding="utf-8")
    else:
        need(path.read_text(encoding="utf-8") == expected, "crosswalk official wording/day drift")
    learner_and_checks()
    assets()
    link_count = links_and_rights()
    expected_manifest = manifest()
    if args.write_manifest:
        MANIFEST.write_text(expected_manifest, encoding="utf-8")
    else:
        need(MANIFEST.read_text(encoding="utf-8") == expected_manifest, "SHA manifest drift")
    print(f"PASS: 10 × 25-minute lessons; 30 core routes; 20 optional extras + 10 home routes; "
          f"2 fresh checks/8 rubric criteria; 15 exact Year 7 English codes/rows; "
          f"3 A4 SVG/PDF/text aids; {link_count} local links; {len(payload_files())} hashed files")


if __name__ == "__main__":
    main()
