#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only Year 2 supplementary audit. Explicit flags update derived files."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
IMPORT = ROOT.parents[4] / "data/frameworks/acara-v9.json"
SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
SUBJECTS = ("science", "hass", "hpe", "arts")
EXPECTED = {
    "science": set("AC9S2U02 AC9S2U03 AC9S2H01 AC9S2I01 AC9S2I02 AC9S2I03 AC9S2I04 AC9S2I05 AC9S2I06".split()),
    "hass": set("AC9HS2K02 AC9HS2K03 AC9HS2S01 AC9HS2S02 AC9HS2S03 AC9HS2S04 AC9HS2S05 AC9HS2S06".split()),
    "hpe": set("AC9HP2P02 AC9HP2P03 AC9HP2P04 AC9HP2P05 AC9HP2M01 AC9HP2M02 AC9HP2M04 AC9HP2M05".split()),
    "arts": set("AC9ADA2D01 AC9ADA2C01 AC9ADA2P01 AC9ADR2D01 AC9ADR2C01 AC9ADR2P01 AC9AMA2D01 AC9AMA2C01 AC9AMA2P01 AC9AMU2D01 AC9AMU2C01 AC9AMU2P01 AC9AVA2D01 AC9AVA2C01 AC9AVA2P01".split()),
}
AREAS = {
    "science": ("Science", "Year 2"),
    "hass": ("Humanities and Social Sciences", "Year 2"),
    "hpe": ("Health and Physical Education", "Years 1 and 2"),
    "arts": ("The Arts", "Years 1 and 2"),
}


def require(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def norm(value: str) -> str:
    return " ".join(value.split())


def source_records() -> tuple[dict, dict[str, dict]]:
    require(IMPORT.is_file(), f"Pinned official import missing: {IMPORT}")
    data = json.loads(IMPORT.read_text(encoding="utf-8"))
    require(data["source_sha256"] == SHA, "Official workbook SHA drift")
    require("australiancurriculum.edu.au" in data["source_url"], "Official workbook URL drift")
    require(data["framework"] == "Australian Curriculum Version 9.0", "Framework drift")
    records = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description"}
    require(len(records) > 1000, "Import lacks content descriptions")
    return data, records


def crosswalk(data: dict, records: dict[str, dict]) -> str:
    rows = [
        "# Australian Curriculum v9 crosswalk · Year 2 supplementary fortnight",
        "",
        "These are **partial links**, not complete coverage or achievement-standard judgments. Science and HASS rows are Year 2; HPE and all five Arts subjects use the official **Years 1 and 2** band. HASS technology facts use a linked museum source, while the invented map only rehearses scale language; actual `AC9HS2K03` evidence requires a checked local map. `AC9HS2K01` and `AC9HS2K04` are deliberately deferred for verified local and community-authorised sources. HPE movement/cooperation, Dance performance, Music sound/listening, and Media Arts technology/presentation are counted only when the named action actually occurs through an accessible route. A paper plan alone cannot supply that evidence.",
        "",
        f"Official source: [ACARA curriculum workbook]({data['source_url']}), accessed {data['retrieved_at']}; source SHA-256 `{data['source_sha256']}`. Wording below is the pinned import's content-description text with whitespace normalised. The source workbook row is retained for audit.",
        "",
        "| Pack | Official code | Source row | Official level | Official subject | Official content description |",
        "|---|---|---:|---|---|---|",
    ]
    for subject in SUBJECTS:
        for code in sorted(EXPECTED[subject]):
            record = records[code]
            a = record["attributes"]
            cells = [subject.upper(), code, str(record["source_row"]), a["level"], a["subject"], norm(record["plain_text"])]
            rows.append("| " + " | ".join(cell.replace("|", "\\|") for cell in cells) + " |")
    rows += [
        "",
        "© Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. Downloaded from the Australian Curriculum website (accessed 29 September 2026) and modified for plain-text display. Curriculum material is licensed under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/). [Terms and exclusions](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse SubjectNest, and SubjectNest is not affiliated with, sponsored or approved by ACARA.",
        "",
        "The pinned import is a dated snapshot, not live synchronisation. Check the current official source and local state/territory implementation before future reissue. See [source, rights and review ledger](SOURCES-AND-REVIEW.md).",
        "",
    ]
    return "\n".join(rows)


def check_content(records: dict[str, dict]) -> None:
    total = 0
    for subject in SUBJECTS:
        teacher = (ROOT / subject / "TEACHER.md").read_text(encoding="utf-8")
        learner = (ROOT / subject / "LEARNER.md").read_text(encoding="utf-8")
        assess = (ROOT / subject / "ASSESSMENT.md").read_text(encoding="utf-8")
        declared_lines = re.findall(r"^\*\*Codes:\*\* ([^\n]+)", teacher, flags=re.M)
        require(len(declared_lines) == 10, f"{subject}: each day needs a code declaration")
        codes = set(re.findall(r"\bAC9[A-Z0-9]+\b", " ".join(declared_lines)))
        require(codes == EXPECTED[subject], f"{subject}: code set mismatch {codes ^ EXPECTED[subject]}")
        area, level = AREAS[subject]
        for code in codes:
            require(code in records, f"Unknown official code {code}")
            a = records[code]["attributes"]
            require(a["learning_area"] == area and a["level"] == level, f"{code}: area/level mismatch")
        for label, doc in (("teacher", teacher), ("learner", learner)):
            days = [int(n) for n in re.findall(r"^## Day (\d+)\b", doc, flags=re.M)]
            require(days == list(range(1, 11)), f"{subject}/{label}: expected Days 1–10 exactly")
        sections = re.split(r"^## Day \d+\b", teacher, flags=re.M)[1:]
        require(len(sections) == 10, f"{subject}: teacher sections")
        for day, section in enumerate(sections, 1):
            phases = re.findall(r"^([123])\. \*\*[^*]+ · (\d+) min\.\*\*", section, flags=re.M)
            require(phases == [("1", "2"), ("2", "7"), ("3", "3")], f"{subject} Day {day}: 2+7+3 phases")
            require("**Home/low-material:**" in section, f"{subject} Day {day}: missing home/low-material route")
        keyed = [int(n) for n in re.findall(r"^\| (\d+)\b", assess, flags=re.M)]
        require(keyed == list(range(1, 11)), f"{subject}: key must cover Days 1–10")
        require("next move" in assess.lower() or "recheck" in assess.lower(), f"{subject}: missing next move")
        require(not re.search(r"ASSESSMENT\.md|\*\*Answer key\*\*", learner, flags=re.I), f"{subject}: learner key leakage")
        total += len(sections)
    require(total == 40, "Expected 40 short blocks")
    arts = (ROOT / "arts/TEACHER.md").read_text(encoding="utf-8")
    for form in ("Dance", "Drama", "Media Arts", "Music", "Visual Arts"):
        require(len(re.findall(rf"^## Day \d+ — {form}:", arts, flags=re.M)) == 2, f"Arts: expected two {form} days")


def check_links() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        content = path.read_text(encoding="utf-8")
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", content):
            if url.startswith(("http://", "https://", "mailto:", "#")):
                continue
            local = (path.parent / unquote(url.split("#", 1)[0])).resolve()
            require(local.is_file(), f"Broken link: {path.relative_to(ROOT)} -> {url}")
            count += 1
    return count


def check_assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    alternatives = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    subprocess.run([sys.executable, str(ROOT / "print/generate_print.py")], check=True, capture_output=True, text=True)
    for subject in SUBJECTS:
        svg = ROOT / "print" / f"{subject}-aid.svg"
        pdf = svg.with_suffix(".pdf")
        require(svg.is_file() and pdf.is_file(), f"{subject}: missing SVG/PDF")
        root = ET.parse(svg).getroot()
        require(root.attrib.get("width") == "210mm" and root.attrib.get("height") == "297mm", f"{subject}: SVG not A4")
        require(root.find("s:title", ns) is not None and root.find("s:desc", ns) is not None, f"{subject}: missing SVG access text")
        info = subprocess.run(["pdfinfo", str(pdf)], capture_output=True, text=True, check=True).stdout
        require("(A4)" in info and "Pages:           1" in info, f"{subject}: PDF not one A4 page")
        searchable = subprocess.run(["pdftotext", str(pdf), "-"], capture_output=True, text=True, check=True).stdout
        require(len(searchable) > 200 and "CC BY 4.0" in searchable, f"{subject}: PDF text/credit missing")
        fonts = subprocess.run(["pdffonts", str(pdf)], capture_output=True, text=True, check=True).stdout
        require("DejaVu" in fonts, f"{subject}: expected embedded DejaVu subset")
        require(re.search(rf"^## (?:The )?{subject}\b", alternatives, flags=re.I | re.M) is not None, f"{subject}: text alternative missing")
    require(re.search(r"ASSESSMENT\.md|\| Day \| Exact check|^Correct:", alternatives, flags=re.I | re.M) is None, "Print alternative leaks key")


def file_hashes() -> str:
    files = sorted(p for p in ROOT.rglob("*") if p.is_file() and p.name != "MANIFEST.sha256" and "__pycache__" not in p.parts)
    return "".join(f"{hashlib.sha256(p.read_bytes()).hexdigest()}  {p.relative_to(ROOT)}\n" for p in files)


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    data, records = source_records()
    expected_crosswalk = crosswalk(data, records)
    crosswalk_path = ROOT / "CURRICULUM-CROSSWALK.md"
    if args.write_crosswalk:
        crosswalk_path.write_text(expected_crosswalk, encoding="utf-8")
    require(crosswalk_path.read_text(encoding="utf-8") == expected_crosswalk, "Crosswalk differs from pinned official import")
    check_content(records)
    links = check_links()
    check_assets()
    expected_manifest = file_hashes()
    manifest = ROOT / "MANIFEST.sha256"
    if args.write_manifest:
        manifest.write_text(expected_manifest, encoding="utf-8")
    require(manifest.read_text(encoding="utf-8") == expected_manifest, "Hash manifest mismatch")
    print(f"PASS: 4 subjects · 40 blocks · {sum(map(len, EXPECTED.values()))} pinned codes · 4 SVG/PDF aids · {links} local links · {len(expected_manifest.splitlines())} hashes")


if __name__ == "__main__":
    try:
        main()
    except (AssertionError, KeyError, FileNotFoundError, subprocess.CalledProcessError) as exc:
        print(f"FAIL: {exc}", file=sys.stderr)
        raise SystemExit(1)
