#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only checks for the Year 6 English pack; writes only by explicit flag."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
ACARA = ROOT.parents[4] / "data/frameworks/acara-v9.json"
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
# Exact Year 6 English content-description rows from the pinned official workbook.
ROWS = {
    "AC9E6LA02": 722, "AC9E6LA03": 727, "AC9E6LA04": 732,
    "AC9E6LA05": 736, "AC9E6LA06": 739, "AC9E6LA07": 743,
    "AC9E6LA09": 751, "AC9E6LY01": 781, "AC9E6LY02": 786,
    "AC9E6LY03": 791, "AC9E6LY05": 797, "AC9E6LY06": 805,
    "AC9E6LY07": 810,
}
DESCRIPTIONS = {
    "AC9E6LA02": "understand the uses of objective and subjective language, and identify bias",
    "AC9E6LA03": "explain how texts across the curriculum are typically organised into characteristic stages and phases depending on purposes, recognising how authors often adapt text structures and language features",
    "AC9E6LA04": "understand that cohesion can be created by the intentional use of repetition, and the use of word associations",
    "AC9E6LA05": "understand how embedded clauses can expand the variety of complex sentences to elaborate, extend and explain ideas",
    "AC9E6LA06": "understand how ideas can be expanded and sharpened through careful choice of verbs, elaborated tenses and a range of adverb groups",
    "AC9E6LA07": "identify and explain how images, figures, tables, diagrams, maps and graphs contribute to meaning",
    "AC9E6LA09": "understand how to use the comma for lists, to separate a dependent clause from an independent clause, and in dialogue",
    "AC9E6LY01": "examine texts including media texts that represent ideas and events, and identify how they reflect the context in which they were created",
    "AC9E6LY02": "use interaction skills and awareness of formality when paraphrasing, questioning, clarifying and interrogating ideas, developing and supporting arguments, and sharing and evaluating information, experiences and opinions",
    "AC9E6LY03": "analyse how text structures and language features work together to meet the purpose of a text, and engage and influence audiences",
    "AC9E6LY05": "use comprehension strategies such as visualising, predicting, connecting, summarising, monitoring and questioning to build literal and inferred meaning, and to connect and compare content from a variety of sources",
    "AC9E6LY06": "plan, create, edit and publish written and multimodal texts whose purposes may be imaginative, informative and persuasive, using paragraphs, a variety of complex sentences, expanded verb groups, tense, topic-specific and vivid vocabulary, punctuation, spelling and visual features",
    "AC9E6LY07": "plan, create, rehearse and deliver spoken and multimodal presentations that include information, arguments and details that develop a theme or idea, organising ideas using precise topic-specific and technical vocabulary, pitch, tone, pace, volume, and visual and digital features",
}
SLICES = {
    "AC9E6LA02": "Days 12, 14: subjective headline and bounded wording; this does not teach every possible bias form.",
    "AC9E6LA03": "Days 16–17: stages of a short persuasive decision note; other genres remain for later weeks.",
    "AC9E6LA04": "Day 19: intentional repetition of a proposal phrase for cohesion; word associations are not fully treated.",
    "AC9E6LA05": "Day 18: one restrictive embedded `that` clause; broader complex-sentence repertoire remains open.",
    "AC9E6LA06": "Days 18, 20: reporting verbs and modal strength; elaborated tense and adverb groups need further teaching.",
    "AC9E6LA07": "Days 12, 15: poster hierarchy and tally tables; diagrams, maps and graphs need later examples.",
    "AC9E6LA09": "Days 18, 20: comma after a fronted dependent clause; list and dialogue uses are not assessed here.",
    "AC9E6LY01": "Days 11, 13, 15, 20: creator, audience and creation context of invented media texts.",
    "AC9E6LY02": "Days 13–14, 16, 19–20: source-linked paraphrase, neutral questions and bounded arguments.",
    "AC9E6LY03": "Days 12, 17: how hierarchy and stages influence readers in two invented forms.",
    "AC9E6LY05": "Days 11, 13–15: predict, summarise, question and compare invented sources.",
    "AC9E6LY06": "Days 16–20: plan, draft, revise and share a short persuasive note; wider purposes remain for the year.",
    "AC9E6LY07": "Day 19: rehearse and optionally deliver a private short pitch; cue-card planning alone is not delivery evidence.",
}
REQUIRED = {
    "README.md", "LESSONS.md", "LEARNER.md", "SOURCE-TEXTS.md", "DAILY-CHOICES.md",
    "DAILY-EXTRAS.md", "STUDENT-CHECKS.md", "teacher/ANSWER-AND-NEXT.md",
    "CURRICULUM-CROSSWALK.md", "SOURCES-AND-REVIEW.md", "RUN-THROUGH.md",
    "CODE-LICENSE.txt", "verify_pack.py", "print/generate_print.py",
    "print/TEXT-ALTERNATIVES.md", "print/FONT-RIGHTS.md",
    "print/dejavu-font-copyright.txt",
}
for stem in ("repair-poster", "source-comparison", "argument-planner"):
    REQUIRED |= {f"print/{stem}.svg", f"print/{stem}.pdf"}


def require(ok: bool, detail: str) -> None:
    if not ok:
        raise AssertionError(detail)


def records() -> tuple[dict, dict[str, dict]]:
    data = json.loads(ACARA.read_text(encoding="utf-8"))
    require(data["source_sha256"] == SOURCE_SHA, "pinned workbook SHA changed")
    require(data["framework"] == "Australian Curriculum Version 9.0", "framework changed")
    require(data["source_url"].startswith("https://www.australiancurriculum.edu.au/"), "official URL changed")
    found = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description" and r.get("code") in ROWS}
    require(set(found) == set(ROWS), "Year 6 English code set missing a pinned row")
    for code, row in ROWS.items():
        r = found[code]
        require(r["source_row"] == row, f"{code}: workbook row drift")
        require((r["attributes"]["level"], r["attributes"]["learning_area"], r["attributes"]["subject"]) ==
                ("Year 6", "English", "English"), f"{code}: official level/area/subject mismatch")
        require(r["plain_text"] == DESCRIPTIONS[code], f"{code}: official description wording drift")
    return data, found


def crosswalk(data: dict, found: dict[str, dict]) -> str:
    out = ["# Australian Curriculum v9 crosswalk · Year 6 English Weeks 3–4", "",
           "These are **partial content-description links** for ten short lessons, not the full Year 6 English curriculum, an achievement-standard judgement or a state/territory approval. The descriptions below are copied from the pinned official workbook import with plain-text whitespace normalisation. Each taught slice names what this pack actually addresses and what needs further teaching.", "",
           f"Source: [official ACARA v9 workbook]({data['source_url']}), retrieved {data['retrieved_at']}; workbook SHA-256 `{data['source_sha256']}`. The source-row numbers allow exact matching against that snapshot.", "",
           "| Official code | Workbook row | Level / area | Official content description | Taught slice and limit |",
           "| --- | ---: | --- | --- | --- |"]
    for code in ROWS:
        r = found[code]
        cells = [code, str(r["source_row"]), "Year 6 / English", " ".join(r["plain_text"].split()), SLICES[code]]
        out.append("| " + " | ".join(c.replace("|", "\\|") for c in cells) + " |")
    out += ["",
            "© Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. Downloaded from the Australian Curriculum website (accessed 29 September 2026) and modified for plain-text display. Curriculum material is licensed under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/). [Terms and exclusions](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse SubjectNest; SubjectNest is not affiliated with, sponsored or approved by ACARA.", "",
            "This is a dated source snapshot, not live synchronisation. Check the current official source and each state/territory implementation before reissue. See [sources, rights and review](SOURCES-AND-REVIEW.md).", ""]
    return "\n".join(out)


def markdown_links() -> None:
    for md in ROOT.rglob("*.md"):
        body = md.read_text(encoding="utf-8")
        for raw in re.findall(r"\]\(([^)]+)\)", body):
            target = unquote(raw.split("#", 1)[0])
            if not target or target.startswith(("https://", "http://", "mailto:")):
                continue
            require((md.parent / target).is_file(), f"Broken local link in {md.relative_to(ROOT)}: {raw}")


def content_checks() -> None:
    for item in REQUIRED:
        require((ROOT / item).is_file(), f"Missing pack file: {item}")
    lesson = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    learner = (ROOT / "LEARNER.md").read_text(encoding="utf-8")
    choices = (ROOT / "DAILY-CHOICES.md").read_text(encoding="utf-8")
    extras = (ROOT / "DAILY-EXTRAS.md").read_text(encoding="utf-8")
    key = (ROOT / "teacher/ANSWER-AND-NEXT.md").read_text(encoding="utf-8")
    checks = (ROOT / "STUDENT-CHECKS.md").read_text(encoding="utf-8")
    source = (ROOT / "SOURCE-TEXTS.md").read_text(encoding="utf-8")
    for name, body, marker in (("lessons", lesson, r"^### Day (\d+) ·"),
                               ("learner", learner, r"^## Day (\d+) ·"),
                               ("choices", choices, r"^## Day (\d+) ·"),
                               ("extras", extras, r"^\| (\d+) \|")):
        days = [int(x) for x in re.findall(marker, body, flags=re.MULTILINE)]
        require(days == list(range(11, 21)), f"{name}: need exactly Days 11–20 in order")
    sections = re.split(r"^## Day \d+ ·", choices, flags=re.MULTILINE)[1:]
    for day, section in enumerate(sections, 11):
        routes = re.findall(r"^- \*\*([ABC]) ·", section, flags=re.MULTILINE)
        require(routes == ["A", "B", "C"], f"Day {day}: three ordered practice choices")
        for route in routes:
            require(f"D{day}-{route}" in key, f"D{day}-{route}: teacher response missing")
    lesson_sections = re.split(r"^### Day \d+ ·", lesson, flags=re.MULTILINE)[1:]
    for day, section in enumerate(lesson_sections, 11):
        for stage, minutes in (("Invite", 2), ("Model", 5), ("Guided", 6), ("Note", 2)):
            require(f"**{stage} · {minutes} min.**" in section, f"Day {day}: {stage} timing")
        if day in (15, 20):
            require("**Independent check · 6 min.**" in section and "**Check exit · 4 min.**" in section, f"Day {day}: fresh-check timing")
            require("**Choice · 6 min.**" not in section, f"Day {day}: later practice must not replace check")
        else:
            require("**Choice · 6 min.**" in section and "**Exit · 4 min.**" in section, f"Day {day}: practice/exit timing")
        require("**Home/extension:**" in section and "**Codes:**" in section, f"Day {day}: missing extension or codes")
        day_codes = re.findall(r"AC9E6[A-Z0-9]+", section)
        require(day_codes and set(day_codes) <= set(ROWS), f"Day {day}: unsupported code")
    require(set(re.findall(r"AC9E6[A-Z0-9]+", lesson)) == set(ROWS), "crosswalk codes and taught lesson codes differ")
    require(len(re.findall(r"^\d\. \*\*", checks, flags=re.MULTILINE)) == 7, "fresh checks need four plus three items")
    require("## Day 15 held-out check" in key and "## Day 20 held-out check" in key, "separate check key missing")
    for marker in ("A KIT FOR EVERYONE!", "pop-up reading cart", "Borrow Box", "19 of 30 guide cards"):
        require(marker not in lesson[:lesson.index("### Day 15")], f"held-out detail in precheck lessons: {marker}")
        require(marker not in source and marker not in choices, f"held-out detail in practice/source packet: {marker}")
    require("11 + 4 + 3 = 18" in lesson and "18 of 30" in key and "9 + 3 = 12" in key,
            "fictional arithmetic audit examples missing")
    require("everything in this packet is invented" in source.lower(), "fictional source disclosure absent")
    require("Later practice" in choices, "held-out day practice not clearly deferred")
    markdown_links()


def print_checks() -> None:
    subprocess.run([sys.executable, str(ROOT / "print/generate_print.py")], check=True, capture_output=True, text=True)
    for stem in ("repair-poster", "source-comparison", "argument-planner"):
        svg = ROOT / f"print/{stem}.svg"
        pdf = ROOT / f"print/{stem}.pdf"
        root = ET.parse(svg).getroot()
        require(root.attrib.get("width") == "210mm" and root.attrib.get("height") == "297mm" and
                root.attrib.get("viewBox") == "0 0 210 297", f"{stem}: SVG A4 dimensions")
        ns = "{http://www.w3.org/2000/svg}"
        require(root.find(ns + "title") is not None and root.find(ns + "desc") is not None,
                f"{stem}: SVG accessible title/description")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        require("Pages:           1" in info and "Page size:       595.276 x 841.89 pts (A4)" in info,
                f"{stem}: PDF not one-page A4")
        copied = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        require(len(copied.strip()) > 200 and "SubjectNest" in copied, f"{stem}: PDF text not selectable")
        fonts = subprocess.check_output(["pdffonts", str(pdf)], text=True)
        require("DejaVu" in fonts, f"{stem}: embedded font drift")
    poster = subprocess.check_output(["pdftotext", str(ROOT / "print/repair-poster.pdf"), "-"], text=True)
    for word in ("ELEVEN", "Four items were waiting for parts", "three were not repaired", "how long any repair lasted"):
        require(word in poster, f"poster missing exact content: {word}")


def digest_manifest() -> str:
    out = []
    for path in sorted(ROOT.rglob("*")):
        if path.is_file() and path.name != "MANIFEST.sha256" and "__pycache__" not in path.parts:
            out.append(f"{hashlib.sha256(path.read_bytes()).hexdigest()}  {path.relative_to(ROOT).as_posix()}")
    return "\n".join(out) + "\n"


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    data, found = records()
    expected = crosswalk(data, found)
    path = ROOT / "CURRICULUM-CROSSWALK.md"
    if args.write_crosswalk:
        path.write_text(expected, encoding="utf-8")
    else:
        require(path.is_file() and path.read_text(encoding="utf-8") == expected, "crosswalk missing or differs from pinned rows")
    content_checks()
    print_checks()
    manifest = ROOT / "MANIFEST.sha256"
    wanted = digest_manifest()
    if args.write_manifest:
        manifest.write_text(wanted, encoding="utf-8")
    else:
        require(manifest.is_file() and manifest.read_text(encoding="utf-8") == wanted, "hash manifest missing or stale")
    print("PASS: 13 exact ACARA Year 6 English rows; 10 x 25-minute lessons; 30 choices; "
          "2 fresh checks/7 items; 3 accessible A4 aid pairs; local links; hashes")


if __name__ == "__main__":
    try:
        main()
    except (AssertionError, subprocess.CalledProcessError, KeyError, ValueError) as exc:
        print(f"FAIL: {exc}", file=sys.stderr)
        raise SystemExit(1)
