#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only source, arithmetic, structure, asset and hash audit for this pack."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
IMPORT = STUDIO / "data/frameworks/acara-v9.json"
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = ["AC9M5N06", "AC9M5N08", "AC9M5N09", "AC9M5A01", "AC9M5A02"]
EVIDENCE = {
    "AC9M5N06": "Days 11–17 and 19–20 multiply larger numbers by one/two digits, compare strategies and reasonableness; limited 2-week number set.",
    "AC9M5N08": "Days 11, 14–17 and 19–20 use prior estimates and bounds including fictional costs; some decisions still need exact totals.",
    "AC9M5N09": "Days 16 and 19–20 formulate fictional product/limit decisions and state unknown information; no full modelling range.",
    "AC9M5A01": "Days 17–18 and 20 use a product/inverse family to check group relationships; division strategies need later practice.",
    "AC9M5A02": "Day 18 solves three multiplication equations with an unknown and substitution; one-day sample only.",
}
CORE = {
    11: [(52, 18, 936), (73, 14, 1022), (208, 5, 1040)],
    12: [(235, 7, 1645), (602, 4, 2408), (481, 6, 2886)],
    13: [(215, 14, 3010), (76, 32, 2432), (142, 26, 3692)],
    14: [(396, 9, 3564), (125, 24, 3000), (318, 19, 6042)],
    15: [(106, 13, 1378), (72, 28, 2016), (309, 6, 1854)],
    16: [(12, 34, 408), (19, 27, 513), (11, 42, 462)],
    17: [(215, 9, 1935), (324, 7, 2268), (507, 6, 3042)],
    18: [(18, 39, 702), (26, 32, 832), (45, 28, 1260)],
    19: [(30, 24, 720), (40, 21, 840), (36, 26, 936)],
    20: [(57, 19, 1083), (24, 35, 840), (108, 11, 1188)],
}
SECOND_OPTIONS = [(45, 16, 720), (30, 28, 840), (24, 39, 936)]
EXTRA_PRODUCTS = [
    (39, 21, 819), (62, 15, 930), (504, 3, 1512), (54, 3, 162), (706, 4, 2824),
    (84, 27, 2268), (46, 32, 1472), (214, 19, 4066), (25, 48, 1200),
    (132, 14, 1848), (119, 20, 2380), (14, 29, 406), (12, 38, 456),
    (306, 8, 2448), (390, 7, 2730), (50, 18, 900), (24, 42, 1008),
    (12, 50, 600), (20, 30, 600), (28, 15, 420), (21, 20, 420),
]
FRESH = [(213, 14, 2982), (178, 24, 4272)]


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def norm(value: str) -> str:
    return " ".join(value.split())


def source() -> tuple[dict, dict[str, dict]]:
    data = json.loads(IMPORT.read_text(encoding="utf-8"))
    need(data["source_sha256"] == SOURCE_SHA and data["framework"] == "Australian Curriculum Version 9.0", "Pinned source/version drift")
    original = STUDIO / "research/sources" / data["source_file"]
    need(original.is_file() and hashlib.sha256(original.read_bytes()).hexdigest() == SOURCE_SHA, "Official workbook bytes drift")
    need(data["source_url"].startswith("https://www.australiancurriculum.edu.au/"), "Nonofficial source URL")
    records = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description" and r.get("code")}
    for code in CODES:
        need(code in records, f"Unknown source code {code}")
        r = records[code]
        need(r["attributes"]["level"] == "Year 5" and r["attributes"]["learning_area"] == "Mathematics" and r["attributes"]["subject"] == "Mathematics", f"Wrong level/subject {code}")
        need(isinstance(r["source_row"], int), f"Missing workbook row {code}")
    return data, records


def crosswalk(data: dict, records: dict[str, dict]) -> str:
    lines = [
        "# Australian Curriculum v9 · Year 5 mathematics Weeks 3–4 crosswalk", "",
        "These are **partial teaching opportunities across ten 25-minute lessons**. They do not mean a content description or achievement standard is fully met, a term order is mandated, or every state/territory uses the same reporting schedule.", "",
        f"Official source: [ACARA curriculum workbook]({data['source_url']}), retrieved {data['retrieved_at']}; original XLSX SHA-256 `{data['source_sha256']}`. Descriptions are exact with whitespace normalised; workbook source rows allow an audit.", "",
        "| Official code | Workbook row | Level | Exact official content description | Taught opportunity and limit |",
        "|---|---:|---|---|---|",
    ]
    for code in CODES:
        r = records[code]
        cells = (code, str(r["source_row"]), "Year 5", norm(r["plain_text"]), EVIDENCE[code])
        lines.append("| " + " | ".join(c.replace("|", "\\|") for c in cells) + " |")
    lines += [
        "", "© Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. Curriculum wording was downloaded from the [Australian Curriculum website](https://www.australiancurriculum.edu.au/downloads) (accessed 29 September 2026), with whitespace/plain-text normalisation, under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/) subject to [terms and exclusions](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse SubjectNest.",
        "", "Original opportunity notes and table arrangement © NeuroForgeIO Pty Ltd 2026, SubjectNest, CC BY 4.0. See [source/rights ledger](SOURCES-AND-REVIEW.md). Recheck official and local sources before reissue.", "",
    ]
    return "\n".join(lines)


def slug(heading: str) -> str:
    return "".join(c for c in heading.lower() if c.isalnum() or c in " -_").replace(" ", "-")


def links_rights() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        body = path.read_text(encoding="utf-8")
        need("CC BY 4.0" in body, f"Rights notice missing: {path.relative_to(ROOT)}")
        for target in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", body):
            if target.startswith(("https://", "http://", "mailto:")):
                continue
            local, _, anchor = unquote(target).partition("#")
            dest = (path.parent / local).resolve() if local else path
            need(dest.is_file(), f"Broken link {path.relative_to(ROOT)} -> {target}")
            if anchor and dest.suffix == ".md":
                heads = re.findall(r"^#{1,6}\s+(.+?)\s*$", dest.read_text(encoding="utf-8"), re.MULTILINE)
                need(anchor in {slug(h) for h in heads}, f"Broken heading anchor {path.relative_to(ROOT)} -> {target}")
            count += 1
    return count


def content_arithmetic() -> None:
    scripts = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    cards = (ROOT / "STUDENT-CARDS.md").read_text(encoding="utf-8")
    extras = (ROOT / "DAILY-EXTRAS.md").read_text(encoding="utf-8")
    key = (ROOT / "teacher/TEACHER-KEY.md").read_text(encoding="utf-8")
    checks = (ROOT / "STUDENT-CHECKS.md").read_text(encoding="utf-8")
    for name, body in (("LESSONS", scripts), ("CARDS", cards), ("EXTRAS", extras)):
        days = [int(v) for v in re.findall(r"^## Day (\d+)\b", body, re.MULTILINE)]
        need(days == list(range(11, 21)), f"{name}: expected Days 11–20 once: {days}")
    blocks = re.split(r"^## Day \d+\b", scripts, flags=re.MULTILINE)[1:]
    for day, block in enumerate(blocks, 11):
        times = [int(x) for x in re.findall(r"^\d+\. \*\*[^*]+ · (\d+) min\.\*\*", block, re.MULTILINE)]
        need(times == ([2, 3, 3, 12, 3, 2] if day in (15, 20) else [2, 5, 6, 7, 3, 2]), f"Day {day}: phase minutes drift {times}")
        need("AC9M5" in block, f"Day {day}: missing exact code")
    need(set(re.findall(r"\bAC9M5[A-Z0-9]+\b", scripts)) == set(CODES), "Lesson code set differs from crosswalk")
    need(re.findall(r"^- \*\*([ABC]) · ", cards, re.MULTILINE) == list("ABC") * 10, "Exactly 30 learner choices required")
    need(re.findall(r"^- \*\*([AB]) · ", extras, re.MULTILINE) == list("AB") * 10, "Exactly 20 optional routes required")
    for day, options in CORE.items():
        row = next((line for line in key.splitlines() if re.match(rf"^\| {day}(?:,| )", line)), None)
        need(row is not None, f"No worked core key for day {day}")
        cells = row.split("|")[2:5]
        need(len(cells) == 3, f"Three answer cells missing on day {day}")
        for i, (a, b, value) in enumerate(options):
            need(a * b == value, f"False product Day {day}{'ABC'[i]}")
            need(f"{value:,}" in cells[i], f"Key missing {value:,} Day {day}{'ABC'[i]}")
    for a, b, value in SECOND_OPTIONS + EXTRA_PRODUCTS + FRESH:
        need(a * b == value, f"False arithmetic {a}×{b}={value}")
    need(470 - 448 == 22 and 400 - 391 == 9 and 420 - 408 == 12 and 530 - 513 == 17 and 480 - 462 == 18, "Budget difference drift")
    need(4500 - 4272 == 228 and 213 * 14 == 2982 and 178 * 24 == 4272, "Fresh check arithmetic drift")
    for value in ("213", "178", "2,982", "4,272"):
        for name in ("LESSONS.md", "MATERIALS.md", "STUDENT-CARDS.md", "DAILY-EXTRAS.md", "print/TEXT-ALTERNATIVES.md"):
            need(re.search(rf"(?<!\d){re.escape(value)}(?!\d)", (ROOT / name).read_text(encoding="utf-8")) is None, f"Fresh value {value} leaked into {name}")
        need(value in key, f"Fresh answer/input {value} missing in key")
    for value in ("213", "178"):
        need(value in checks, f"Fresh input {value} missing in learner check")
    need("not yet observed" in key.lower(), "Missing evidence handling absent")
    for label in ("Day 15", "Day 20", "Check A", "Check B"):
        need(label in checks and label in key, f"Fresh check/key missing {label}")


def visuals() -> None:
    ns = "{http://www.w3.org/2000/svg}"
    for stem, phrases in {
        "partial-products-mat": ("Partial products", "20 groups", "3 groups", "2 944", "MY NEW PROBLEM"),
        "estimate-exact-decision": ("Estimate", "QUESTION", "BENCHMARK", "EXACT PRODUCT", "LIMIT"),
    }.items():
        svg = ROOT / "print" / f"{stem}.svg"
        pdf = svg.with_suffix(".pdf")
        tree = ET.parse(svg).getroot()
        need(tree.get("viewBox") == "0 0 210 297" and tree.get("role") == "img", f"A4/access metadata drift: {stem}")
        need(tree.find(ns + "title") is not None and tree.find(ns + "desc") is not None, f"SVG title/desc missing: {stem}")
        need("CC BY 4.0" in svg.read_text(encoding="utf-8"), f"SVG rights missing: {stem}")
        info = subprocess.run(["pdfinfo", str(pdf)], capture_output=True, text=True, check=True).stdout
        need("(A4)" in info and "Pages:           1" in info, f"PDF not one A4 page: {stem}")
        extracted = subprocess.run(["pdftotext", str(pdf), "-"], capture_output=True, text=True, check=True).stdout
        for phrase in phrases:
            need(phrase in extracted, f"PDF text missing {phrase}: {stem}")
        fonts = subprocess.run(["pdffonts", str(pdf)], capture_output=True, text=True, check=True).stdout
        need("DejaVu" in fonts, f"Embedded font changed: {stem}")
    alt = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    for phrase in ("2 560", "384", "2 944", "TENS GROUPS", "BENCHMARK", "LIMIT", "Tactile build"):
        need(phrase in alt, f"Text/tactile alternative missing {phrase}")


def hashes() -> str:
    paths = sorted(p for p in ROOT.rglob("*") if p.is_file() and p.name != "MANIFEST.sha256" and "__pycache__" not in p.parts)
    return "".join(f"{hashlib.sha256(p.read_bytes()).hexdigest()}  {p.relative_to(ROOT)}\n" for p in paths)


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    data, records = source()
    expected = crosswalk(data, records)
    crosswalk_path = ROOT / "CURRICULUM-CROSSWALK.md"
    if args.write_crosswalk:
        crosswalk_path.write_text(expected, encoding="utf-8")
    need(crosswalk_path.read_text(encoding="utf-8") == expected, "Pinned crosswalk drift")
    content_arithmetic()
    visuals()
    links = links_rights()
    expected_hashes = hashes()
    manifest = ROOT / "MANIFEST.sha256"
    if args.write_manifest:
        manifest.write_text(expected_hashes, encoding="utf-8")
    need(manifest.read_text(encoding="utf-8") == expected_hashes, "Hash manifest drift")
    print(f"PASS: 10×25-minute lessons · 30 core and 20 optional routes · 2 held-out checks · {len(CODES)} pinned Year 5 maths codes · 2 original A4 SVG/PDF/text aids · {links} local links · {len(expected_hashes.splitlines())} hashes")


if __name__ == "__main__":
    try:
        main()
    except (AssertionError, FileNotFoundError, KeyError, subprocess.CalledProcessError) as error:
        print(f"FAIL: {error}", file=sys.stderr)
        raise SystemExit(1)
