#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only source/content/link/print/hash audit for Year 3 supplement.

Only --write-crosswalk and --write-manifest change derived files. Rendered
lesson pages and SVG/PDF assets have their own opt-in generation commands.
"""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

from build_content import CARDS, LEARNER_PROMPTS, render

ROOT = Path(__file__).resolve().parent
SOURCE = ROOT.parents[4] / "data/frameworks/acara-v9.json"
SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
SUBJECTS = ("science", "hass", "hpe", "technologies", "arts")
EXPECTED = {
    "science": {"AC9S3H01", "AC9S3H02", "AC9S3I01", "AC9S3I02", "AC9S3I03", "AC9S3I04", "AC9S3I05", "AC9S3I06", "AC9S3U02", "AC9S3U03"},
    "hass": {"AC9HS3K06", "AC9HS3K07", "AC9HS3S01", "AC9HS3S02", "AC9HS3S03", "AC9HS3S04", "AC9HS3S05", "AC9HS3S06", "AC9HS3S07"},
    "hpe": {"AC9HP4M01", "AC9HP4M02", "AC9HP4M03", "AC9HP4M07", "AC9HP4M08", "AC9HP4M09", "AC9HP4P04", "AC9HP4P05", "AC9HP4P06", "AC9HP4P07", "AC9HP4P08", "AC9HP4P09"},
    "technologies": {"AC9TDE4K01", "AC9TDE4K02", "AC9TDE4P01", "AC9TDE4P02", "AC9TDE4P03", "AC9TDE4P04", "AC9TDE4P05", "AC9TDI4K03", "AC9TDI4P01", "AC9TDI4P02", "AC9TDI4P03", "AC9TDI4P05"},
    "arts": {"AC9ADA4C01", "AC9ADA4D01", "AC9ADA4P01", "AC9ADR4C01", "AC9ADR4D01", "AC9ADR4P01", "AC9AMA4C01", "AC9AMA4D01", "AC9AMA4P01", "AC9AMU4C01", "AC9AMU4D01", "AC9AMU4P01", "AC9AVA4C01", "AC9AVA4D01", "AC9AVA4P01"},
}
AREA_LEVEL = {
    "science": ("Science", "Year 3"),
    "hass": ("Humanities and Social Sciences", "Year 3"),
    "hpe": ("Health and Physical Education", "Years 3 and 4"),
    "technologies": ("Technologies", "Years 3 and 4"),
    "arts": ("The Arts", "Years 3 and 4"),
}


def require(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def norm(value: str) -> str:
    return " ".join(value.split())


def source_records() -> tuple[dict, dict[str, dict]]:
    data = json.loads(SOURCE.read_text(encoding="utf-8"))
    require(data["source_sha256"] == SHA, "Pinned workbook checksum drift")
    require(data["framework"] == "Australian Curriculum Version 9.0", "Framework drift")
    require("australiancurriculum.edu.au" in data["source_url"], "Official source URL drift")
    records = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description"}
    require(len(records) > 1000, "Import too small")
    return data, records


def crosswalk(data: dict, records: dict[str, dict]) -> str:
    out = [
        "# Australian Curriculum v9 crosswalk · Year 3 supplementary Weeks 3–4",
        "",
        "These are **partial content-description links**, not complete curriculum or achievement-standard coverage. Science and HASS are Year 3. HPE, both Technologies subjects and all five Arts subjects are officially in the **Years 3 and 4** band. The cards exercise selected descriptions; a 12-minute activity alone cannot establish achievement. Real local HASS rules/community claims need a current approved source. No First Nations Country/Place content, cultural forms or translations are invented. Actual action matters: paper movement plans, paper media storyboards and silent beat cards do not show bodily performance, media-tool production or sounded music.",
        "",
        f"Source: [ACARA official workbook]({data['source_url']}), retrieved {data['retrieved_at']}; SHA-256 `{data['source_sha256']}`. Wording below is the pinned import's content-description text with whitespace normalised. Original workbook rows are retained for audit.",
        "",
        "| Pack | Official code | Source row | Official level | Official subject | Official content description |",
        "| --- | --- | ---: | --- | --- | --- |",
    ]
    for subject in SUBJECTS:
        for code in sorted(EXPECTED[subject]):
            r = records[code]
            a = r["attributes"]
            cells = [subject.upper(), code, str(r["source_row"]), a["level"], a["subject"], norm(r["plain_text"])]
            out.append("| " + " | ".join(cell.replace("|", "\\|") for cell in cells) + " |")
    out += [
        "",
        "© Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. Downloaded from the Australian Curriculum website (accessed 29 September 2026) and modified for plain-text display. Curriculum material is licensed under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/). [Terms and exclusions](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse SubjectNest, and SubjectNest is not affiliated with, sponsored or approved by ACARA.",
        "",
        "This is a dated import, not a live update feed. Check the current official curriculum and state or territory implementation before reissue. See [source and rights ledger](SOURCES-AND-REVIEW.md).",
        "",
    ]
    return "\n".join(out)


def check_cards(records: dict[str, dict]) -> None:
    require(tuple(CARDS) == SUBJECTS, "Subject order drift")
    all_codes: set[str] = set()
    for subject in SUBJECTS:
        cards = CARDS[subject]
        require(len(cards) == len(LEARNER_PROMPTS[subject]) == 10, f"{subject}: need ten distinct cards")
        subject_codes = {code for card in cards for code in card.codes.split()}
        require(subject_codes == EXPECTED[subject], f"{subject}: code set differs")
        for code in subject_codes:
            require(code in records, f"Unknown code {code}")
            a = records[code]["attributes"]
            require((a["learning_area"], a["level"]) == AREA_LEVEL[subject], f"{code}: area or band mismatch")
        all_codes |= subject_codes
        for idx, card in enumerate(cards, 11):
            require(all((card.title, card.codes, card.notice, card.try_, card.show, card.choices, card.check, card.next_, card.prepare, card.home)), f"{subject} Day {idx}: empty teaching field")
            require(len(LEARNER_PROMPTS[subject][idx - 11]) >= 110, f"{subject} Day {idx}: thin learner task")
        for name, expected in render(subject).items():
            path = ROOT / subject / name
            require(path.is_file() and path.read_text(encoding="utf-8") == expected, f"{subject}/{name}: generated lesson drift")
        teacher = (ROOT / subject / "TEACHER.md").read_text(encoding="utf-8")
        learner = (ROOT / subject / "LEARNER.md").read_text(encoding="utf-8")
        key = (ROOT / subject / "ASSESSMENT.md").read_text(encoding="utf-8")
        for label, doc in (("teacher", teacher), ("learner", learner)):
            days = [int(n) for n in re.findall(r"^## Day (\d+)\b", doc, flags=re.MULTILINE)]
            require(days == list(range(11, 21)), f"{subject}/{label}: need Days 11–20 exactly")
        require(len(re.findall(r"^1\. \*\*Notice · 2 min\.\*\*", teacher, flags=re.MULTILINE)) == 10, f"{subject}: notice timings")
        require(len(re.findall(r"^2\. \*\*Try/make · 7 min\.\*\*", teacher, flags=re.MULTILINE)) == 10, f"{subject}: making timings")
        require(len(re.findall(r"^3\. \*\*Show · 3 min\.\*\*", teacher, flags=re.MULTILINE)) == 10, f"{subject}: show timings")
        require(len(re.findall(r"^\| (?:1[1-9]|20) \|", key, flags=re.MULTILINE)) == 10, f"{subject}: separate key rows")
        require("ASSESSMENT.md" not in learner and "answer key" not in learner.lower(), f"{subject}: teacher key leakage")
    require(len(all_codes) == 58, "Unique code count drift")
    for form in ("Dance", "Drama", "Media Arts", "Music", "Visual Arts"):
        require(sum(card.title.startswith(form + ":") for card in CARDS["arts"]) == 2, f"Arts: need two {form} cards")
    require(all("MODEL" in LEARNER_PROMPTS["science"][i] for i in (3, 5, 7, 8, 9)), "Science model labels missing")


def check_links() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        content = path.read_text(encoding="utf-8")
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", content):
            if url.startswith(("http://", "https://", "mailto:", "#")):
                continue
            local = (path.parent / unquote(url.split("#", 1)[0])).resolve()
            require(local.is_file(), f"Broken link: {path.relative_to(ROOT)} -> {url}")
            count += 1
    return count


def check_assets() -> None:
    subprocess.run([sys.executable, str(ROOT / "print/generate_print.py")], check=True, capture_output=True, text=True)
    alternatives = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    ns = {"svg": "http://www.w3.org/2000/svg"}
    for subject in SUBJECTS:
        svg = ROOT / "print" / f"{subject}-aid.svg"
        pdf = svg.with_suffix(".pdf")
        require(svg.is_file() and pdf.is_file(), f"{subject}: missing SVG/PDF")
        root = ET.parse(svg).getroot()
        require(root.attrib.get("width") == "210mm" and root.attrib.get("height") == "297mm", f"{subject}: SVG not A4")
        require(root.find("svg:title", ns) is not None and root.find("svg:desc", ns) is not None, f"{subject}: missing SVG title/description")
        info = subprocess.run(["pdfinfo", str(pdf)], check=True, capture_output=True, text=True).stdout
        require("(A4)" in info and "Pages:           1" in info, f"{subject}: PDF not one A4 page")
        words = subprocess.run(["pdftotext", str(pdf), "-"], check=True, capture_output=True, text=True).stdout
        require(len(words) > 250 and "CC BY 4.0" in words, f"{subject}: PDF text/credit missing")
        fonts = subprocess.run(["pdffonts", str(pdf)], check=True, capture_output=True, text=True).stdout
        require("DejaVu" in fonts, f"{subject}: DejaVu subset missing")
        require(re.search(rf"^## {subject}\b", alternatives, flags=re.MULTILINE | re.IGNORECASE), f"{subject}: text equivalent missing")
    require("dejavu-font-copyright.txt" in (ROOT / "print/FONT-RIGHTS.md").read_text(encoding="utf-8"), "Font rights notice")


def hashes() -> str:
    files = sorted(p for p in ROOT.rglob("*") if p.is_file() and p.name != "MANIFEST.sha256" and "__pycache__" not in p.parts)
    return "".join(f"{hashlib.sha256(p.read_bytes()).hexdigest()}  {p.relative_to(ROOT)}\n" for p in files)


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    data, records = source_records()
    crosswalk_text = crosswalk(data, records)
    crosswalk_path = ROOT / "CURRICULUM-CROSSWALK.md"
    if args.write_crosswalk:
        crosswalk_path.write_text(crosswalk_text, encoding="utf-8")
    require(crosswalk_path.read_text(encoding="utf-8") == crosswalk_text, "Crosswalk differs from pinned source")
    check_cards(records)
    links = check_links()
    check_assets()
    expected = hashes()
    manifest = ROOT / "MANIFEST.sha256"
    if args.write_manifest:
        manifest.write_text(expected, encoding="utf-8")
    require(manifest.read_text(encoding="utf-8") == expected, "Hash manifest mismatch")
    print(f"PASS: 5 subjects · 50 distinct 12-minute blocks · 58 pinned codes · 5 original SVG/PDF aids · {links} local links · {len(expected.splitlines())} file hashes")


if __name__ == "__main__":
    try:
        main()
    except (AssertionError, KeyError, FileNotFoundError, subprocess.CalledProcessError) as exc:
        print(f"FAIL: {exc}", file=sys.stderr)
        raise SystemExit(1)
