#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only Year 1 supplementary pack checks; explicit flags refresh derived files."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
IMPORT = ROOT.parents[4] / "data/frameworks/acara-v9.json"
SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
SUBJECTS = ("science", "hass", "hpe", "arts")
EXPECTED = {
    "science": set("AC9S1H01 AC9S1I01 AC9S1I02 AC9S1I03 AC9S1I04 AC9S1I05 AC9S1I06 AC9S1U01 AC9S1U02".split()),
    "hass": set("AC9HS1K01 AC9HS1K02 AC9HS1K03 AC9HS1S01 AC9HS1S02 AC9HS1S03 AC9HS1S04 AC9HS1S05 AC9HS1S06".split()),
    "hpe": set("AC9HP2M01 AC9HP2M02 AC9HP2M03 AC9HP2M04 AC9HP2M05 AC9HP2P02 AC9HP2P03 AC9HP2P04 AC9HP2P05".split()),
    "arts": set("AC9ADA2C01 AC9ADA2D01 AC9ADA2P01 AC9ADR2C01 AC9ADR2D01 AC9ADR2P01 AC9AMA2C01 AC9AMA2D01 AC9AMA2P01 AC9AMU2C01 AC9AMU2D01 AC9AMU2P01 AC9AVA2C01 AC9AVA2D01 AC9AVA2P01".split()),
}
AREAS = {
    "science": ("Science", "Year 1"),
    "hass": ("Humanities and Social Sciences", "Year 1"),
    "hpe": ("Health and Physical Education", "Years 1 and 2"),
    "arts": ("The Arts", "Years 1 and 2"),
}


def require(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def norm(value: str) -> str:
    return " ".join(value.split())


def source_records() -> tuple[dict, dict[str, dict]]:
    require(IMPORT.is_file(), f"Pinned official import missing: {IMPORT}")
    data = json.loads(IMPORT.read_text(encoding="utf-8"))
    require(data["source_sha256"] == SHA, "Official workbook SHA drift")
    require("ac-version-9" in data["source_url"], "Official workbook URL drift")
    require(data["framework"] == "Australian Curriculum Version 9.0", "Framework drift")
    records = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description"}
    require(len(records) > 1000, "Import does not contain curriculum descriptions")
    return data, records


def crosswalk(data: dict, records: dict[str, dict]) -> str:
    rows = [
        "# Australian Curriculum v9 crosswalk · Year 1 supplementary fortnight",
        "",
        "These are **partial links**, not a declaration that a description or achievement standard has been fully taught. Science and HASS are Year 1 descriptions; HPE and The Arts are Years 1 and 2 band descriptions. HASS local-place knowledge needs genuine local evidence beyond fictional feature cards. Both Media Arts creation/development rows need actual suitable approved media technology use. Music listening/playing, dance performance and arts presentation rows require evidence in the named art form; paper plans alone do not show those outcomes. HPE physical movement needs an observed accessible action; a token plan shows reasoning only. Outdoor physical activity is observed only with approved real participation, not a scene-card discussion.",
        "",
        f"Official source: [ACARA curriculum workbook]({data['source_url']}), accessed {data['retrieved_at']}; source SHA-256 `{data['source_sha256']}`. Description wording below has only whitespace normalised for plain-text display. The original workbook's source row is retained for audit.",
        "",
        "| Pack | Official code | Source row | Official level | Official subject | Official content description |",
        "|---|---|---:|---|---|---|",
    ]
    for subject in SUBJECTS:
        for code in sorted(EXPECTED[subject]):
            record = records[code]
            a = record["attributes"]
            cells = [subject.upper(), code, str(record["source_row"]), a["level"], a["subject"], norm(record["plain_text"])]
            rows.append("| " + " | ".join(x.replace("|", "\\|") for x in cells) + " |")
    rows += [
        "",
        "© Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. Downloaded from the Australian Curriculum website (accessed 29 September 2026) and modified for plain-text display. Curriculum material is licensed under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/). [Terms and exclusions](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse SubjectNest, and SubjectNest is not affiliated with, sponsored or approved by ACARA.",
        "",
        "The pinned import is a dated snapshot, not automatic synchronisation. Validate the current ACARA workbook before future publication or a change to teaching content. See [source and rights review](SOURCES-AND-REVIEW.md).",
        "",
    ]
    return "\n".join(rows)


def check_content(records: dict[str, dict]) -> None:
    total = 0
    for subject in SUBJECTS:
        teacher = (ROOT / subject / "TEACHER.md").read_text(encoding="utf-8")
        learner = (ROOT / subject / "LEARNER.md").read_text(encoding="utf-8")
        assess = (ROOT / subject / "ASSESSMENT.md").read_text(encoding="utf-8")
        codes = set(re.findall(r"\bAC9[A-Z0-9]+\b", teacher))
        require(codes == EXPECTED[subject], f"{subject}: code set mismatch: {codes ^ EXPECTED[subject]}")
        area, level = AREAS[subject]
        for code in codes:
            require(code in records, f"{subject}: unknown official code {code}")
            r = records[code]
            require(r["attributes"]["learning_area"] == area, f"{code}: learning area mismatch")
            require(r["attributes"]["level"] == level, f"{code}: level mismatch")
        for label, doc in (("teacher", teacher), ("learner", learner)):
            days = [int(n) for n in re.findall(r"^## Day (\d+)\b", doc, re.M)]
            require(days == list(range(1, 11)), f"{subject}/{label}: expected Days 1–10 exactly")
        day_sections = re.split(r"^## Day \d+\b", teacher, flags=re.M)[1:]
        require(len(day_sections) == 10, f"{subject}: teacher sections")
        for day, section in enumerate(day_sections, 1):
            phases = re.findall(r"^([123])\. \*\*[^*]+ · (\d+) min\.\*\*", section, re.M)
            require(phases == [("1", "2"), ("2", "7"), ("3", "3")], f"{subject} Day {day}: 2+7+3 phases")
            require("**Home/low-material:**" in section, f"{subject} Day {day}: home/low-material route")
        for week in (1, 2):
            require(f"**Week {week}" in teacher, f"{subject}: week {week} real-life examples")
        keyed_days = [int(n) for n in re.findall(r"^\| (\d+) \|", assess, re.M)]
        require(keyed_days == list(range(1, 11)), f"{subject}: teacher keys Days 1–10")
        require("next" in assess.lower() and "rubric" in assess.lower(), f"{subject}: formative next moves and rubric")
        require(not re.search(r"ASSESSMENT\.md|teacher key|answer key|^\*\*Answer", learner, re.I | re.M), f"{subject}: learner key leakage")
        total += len(day_sections)
    require(total == 40, "Expected 40 supplementary activities")


def check_links() -> int:
    found = 0
    for path in ROOT.rglob("*.md"):
        text = path.read_text(encoding="utf-8")
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", text):
            if url.startswith(("http://", "https://", "mailto:")) or url.startswith("#"):
                continue
            local = (path.parent / unquote(url.split("#", 1)[0])).resolve()
            require(local.is_file(), f"Broken link: {path.relative_to(ROOT)} -> {url}")
            found += 1
    return found


def check_assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    alternatives = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    for name in ("science", "hass", "hpe", "arts"):
        svg = ROOT / "print" / f"{name}-aid.svg"
        pdf = svg.with_suffix(".pdf")
        require(svg.is_file() and pdf.is_file(), f"{name}: missing visual aid")
        tree = ET.parse(svg)
        s = tree.getroot()
        require(s.attrib["width"] == "210mm" and s.attrib["height"] == "297mm", f"{name}: SVG A4")
        require(s.find("s:title", ns) is not None and s.find("s:desc", ns) is not None, f"{name}: SVG access metadata")
        info = subprocess.run(["pdfinfo", str(pdf)], capture_output=True, text=True, check=True).stdout
        require("Page size:       595.276 x 841.89 pts (A4)" in info or "(A4)" in info, f"{name}: PDF A4")
        searchable = subprocess.run(["pdftotext", str(pdf), "-"], capture_output=True, text=True, check=True).stdout
        require(len(searchable) > 250 and "CC BY 4.0" in searchable, f"{name}: PDF text/credit")
        fonts = subprocess.run(["pdffonts", str(pdf)], capture_output=True, text=True, check=True).stdout
        require("DejaVu" in fonts, f"{name}: expected embedded DejaVu subset")
        require(f"## {name if name != 'hpe' else 'HPE'}" in alternatives or (name == "arts" and "## The Arts" in alternatives) or (name == "hass" and "## HASS" in alternatives) or (name == "science" and "## Science" in alternatives), f"{name}: text alternative")
    require(not re.search(r"ASSESSMENT\.md|answer key", alternatives, re.I), "Print text leaks assessment key")


def file_hashes() -> str:
    files = sorted(p for p in ROOT.rglob("*") if p.is_file() and p.name != "MANIFEST.sha256" and "__pycache__" not in p.parts and not p.name.startswith("_preview-"))
    return "".join(f"{hashlib.sha256(p.read_bytes()).hexdigest()}  {p.relative_to(ROOT)}\n" for p in files)


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    data, records = source_records()
    expected_crosswalk = crosswalk(data, records)
    crosswalk_path = ROOT / "CURRICULUM-CROSSWALK.md"
    if args.write_crosswalk:
        crosswalk_path.write_text(expected_crosswalk, encoding="utf-8")
    require(crosswalk_path.read_text(encoding="utf-8") == expected_crosswalk, "Crosswalk differs from official import")
    check_content(records)
    links = check_links()
    check_assets()
    manifest = ROOT / "MANIFEST.sha256"
    expected_manifest = file_hashes()
    if args.write_manifest:
        manifest.write_text(expected_manifest, encoding="utf-8")
    require(manifest.read_text(encoding="utf-8") == expected_manifest, "Hash manifest mismatch")
    print(f"PASS: 4 subjects · 40 activities · {sum(map(len, EXPECTED.values()))} official codes · 4 SVG/PDF aids · {links} local links · {len(expected_manifest.splitlines())} hashes")


if __name__ == "__main__":
    try:
        main()
    except (AssertionError, KeyError, FileNotFoundError, subprocess.CalledProcessError) as exc:
        print(f"FAIL: {exc}", file=sys.stderr)
        raise SystemExit(1)
