#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Check the Year 8 authored sequence and optionally refresh its receipt.

Run from anywhere: python3 products/curriculum-studio/content/year-8/verify_pack.py
To update after reviewing edits: add --write-manifest. No network or packages needed.
"""

from __future__ import annotations

import hashlib
import json
import random
import re
import subprocess
import sys
from pathlib import Path
from urllib.parse import unquote
from xml.etree import ElementTree

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[1]
SOURCE = STUDIO / "data/frameworks/acara-v9.json"
WORKBOOK = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
MANIFEST = ROOT / "manifest.json"
WORKBOOK_HASH = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
SOURCE_URL = "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx"
TERMS_URL = "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use"
REQUIRED_CODES = {
    "AC9E8LA03", "AC9E8LA04", "AC9E8LA05", "AC9E8LA06",
    "AC9E8LY02", "AC9E8LY03", "AC9E8LY06", "AC9E8LY07",
    "AC9M8A01", "AC9M8A02", "AC9M8A03",
    "AC9M8ST01", "AC9M8ST02", "AC9M8ST03", "AC9M8ST04",
    "AC9HE8K01", "AC9HE8S01", "AC9HE8S02", "AC9HE8S03", "AC9HE8S04", "AC9HE8S05",
}


def digest(data: bytes) -> str:
    return hashlib.sha256(data).hexdigest()


def assets() -> list[Path]:
    return sorted(p for p in ROOT.rglob("*") if p.is_file() and p != MANIFEST and "__pycache__" not in p.parts)


def rights(path: Path) -> str:
    if path.name in {
        "CURRICULUM-CROSSWALK.md",
        "SOURCE-AND-RIGHTS.md",
        "SOURCES-AND-REVIEW.md",
        "SOURCES-AND-RIGHTS.md",
    }:
        return "Original mapping and notes CC BY 4.0; any quoted ACARA description has separate ACARA terms"
    if path.name == "CODE-LICENSE.txt":
        return "MIT licence notice for original mathematics print generator" if "mathematics" in path.parts else "Apache-2.0 licence notice for original code"
    if path.name == "FONT-RIGHTS.md":
        return "Third-party DejaVu font notice under Bitstream Vera licence; not SubjectNest CC BY"
    if path.name == "dejavu-font-copyright.txt":
        return "Third-party DejaVu/Bitstream Vera licence notice"
    if path.suffix == ".pdf":
        return "Original SubjectNest design CC BY 4.0; embedded DejaVu font subsets under separate FONT-RIGHTS.md licence"
    if path.suffix == ".py":
        if path.name == "generate_print.py" and "mathematics" in path.parts:
            return "Original mathematics print generator under MIT licence in print/CODE-LICENSE.txt"
        return "Original build/verification code Apache-2.0"
    if path.suffix == ".wav":
        return "Original SubjectNest audio CC BY 4.0; text alternative supplied"
    if path.suffix == ".sha256" or path.name == "manifest.json":
        return "Integrity metadata for the authored pack"
    return "CC BY 4.0; original SubjectNest material"


def curriculum_records() -> dict[str, dict]:
    data = json.loads(SOURCE.read_text(encoding="utf-8"))
    assert data["source_sha256"] == WORKBOOK_HASH, "ACARA workbook hash drift"
    assert data["source_url"] == SOURCE_URL, "ACARA workbook URL drift"
    assert data["dataset_version"] == "acara-v9.0-snapshot-2026-09-29", "canonical ACARA snapshot changed"
    assert WORKBOOK.is_file() and digest(WORKBOOK.read_bytes()) == WORKBOOK_HASH, "pinned official workbook missing or changed"
    return {
        r["code"]: r
        for r in data["records"]
        if r["record_type"] == "content_description"
        and r["attributes"].get("level") in {"Year 8", "Years 7 and 8"}
    }


def code_refs() -> set[str]:
    refs: set[str] = set()
    for path in ROOT.rglob("*.md"):
        refs.update(re.findall(r"\bAC9[A-Z0-9]+\b", path.read_text(encoding="utf-8")))
    return refs


def expected_manifest() -> dict:
    source = curriculum_records()
    refs = code_refs()
    assert REQUIRED_CODES <= refs, f"required code not present in source notes: {sorted(REQUIRED_CODES - refs)}"
    assert refs <= source.keys(), f"wrong-level or absent Year 8/band codes: {sorted(refs - source.keys())}"
    return {
        "pack": "SubjectNest Year 8 opening sequence",
        "version": "0.2.0-draft",
        "status": "Weeks 1-4 English/mathematics/integrated scripted; Weeks 3-4 five-area optional blocks; Weeks 5-40 plan-only; local educator, access, privacy and jurisdiction review pending",
        "original_author": "NeuroForgeIO Pty Ltd",
        "curriculum_source": {
            "publisher": "ACARA",
            "framework": "Australian Curriculum Version 9.0",
            "source_url": SOURCE_URL,
            "retrieved_at": "2026-09-29",
            "workbook_sha256": WORKBOOK_HASH,
            "terms_url": TERMS_URL,
            "not_endorsed_by_acara": True,
        },
        "codes": {
            code: {
                "level": source[code]["attributes"]["level"],
                "source_row": source[code]["source_row"],
                "learning_area": source[code]["attributes"].get("learning_area", ""),
                "description_sha256": digest(source[code]["plain_text"].encode("utf-8")),
            }
            for code in sorted(refs)
        },
        "files": [
            {
                "path": path.relative_to(ROOT).as_posix(),
                "sha256": digest(path.read_bytes()),
                "rights": rights(path),
            }
            for path in assets()
        ],
        "rights_note": "Original lessons, fictional data, print designs and visuals CC BY 4.0 with credit, source, licence and change notice; build/verification code Apache-2.0 except the Year 8 mathematics print generator, which is MIT under its own CODE-LICENSE.txt. Curriculum remains under ACARA terms. Embedded PDF DejaVu font subsets retain the separate Bitstream Vera licence in FONT-RIGHTS.md. No third-party audio, image or translated assets included.",
    }


def check_lesson_minutes(path: Path, days: list[int], minutes: int) -> None:
    content = path.read_text(encoding="utf-8")
    headers = list(re.finditer(r"^## Day (\d+)\b", content, re.MULTILINE))
    actual = [int(h.group(1)) for h in headers]
    assert actual == days, f"{path}: day order {actual}, expected {days}"
    for i, header in enumerate(headers):
        end = headers[i + 1].start() if i + 1 < len(headers) else len(content)
        block = content[header.end():end]
        steps = [int(x) for x in re.findall(r"^\d+\. \*\*[^\n*]*? · (\d+) min\.\*\*", block, re.MULTILINE)]
        assert len(steps) == 6 and sum(steps) == minutes, (path, actual[i], steps)
        assert "**Success:**" in block or minutes == 35, (path, actual[i], "missing success")


def check_links() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        for target in re.findall(r"\[[^\]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if re.match(r"^[a-z]+://", target) or target.startswith("mailto:"):
                continue
            local, _, anchor = unquote(target).partition("#")
            resolved = (path.parent / local).resolve() if local else path.resolve()
            assert resolved.is_file() and ROOT in resolved.parents, f"broken or out-of-pack link: {path}: {target}"
            if anchor and resolved.suffix.lower() == ".md":
                text = resolved.read_text(encoding="utf-8")
                headings = re.findall(r"^#{1,6} +(.+?) *$", text, re.MULTILINE)
                slug_set = {
                    re.sub(r"\s+", "-", re.sub(r"[^\w\s-]", "", re.sub(r"<[^>]+>", "", h).lower())).strip("-")
                    for h in headings
                }
                assert anchor in slug_set, f"broken anchor: {path}: {target}"
            count += 1
    return count


def check_print() -> tuple[int, int]:
    opening_print = ROOT / "term-1/weeks-01-02/print"
    svgs = list(opening_print.glob("*.svg"))
    pdfs = list(opening_print.glob("*.pdf"))
    assert len(svgs) == len(pdfs) == 3, "expected three paired editable/print aids"
    font_notice = ROOT / "term-1/weeks-01-02/print/FONT-RIGHTS.md"
    assert "Bitstream Vera" in font_notice.read_text(encoding="utf-8"), "missing embedded-font rights notice"
    for svg in svgs:
        element = ElementTree.parse(svg).getroot()
        ns = "{http://www.w3.org/2000/svg}"
        assert element.get("viewBox") and element.get("role") == "img", svg
        assert element.get("width") == "297mm" and element.get("height") == "210mm", svg
        assert element.get("aria-labelledby") == "title desc", svg
        assert element.find(f"{ns}title") is not None and element.find(f"{ns}desc") is not None, svg
        assert "CC BY 4.0" in svg.read_text(encoding="utf-8"), svg
        assert svg.with_suffix(".pdf").read_bytes().startswith(b"%PDF"), svg
    alternatives = (ROOT / "term-1/weeks-01-02/print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    for phrase in ("(6,60)", "3/20", "16/20", "What this cannot show"):
        assert phrase in alternatives, f"missing text equivalent: {phrase}"
    return len(svgs), len(pdfs)


def check_later_subpacks() -> None:
    for relative in (
        "english/term-1/weeks-03-04",
        "mathematics/term-1/weeks-03-04",
        "integrated/term-1/weeks-03-04",
        "supplementary/term-1/weeks-03-04",
    ):
        verifier = ROOT / relative / "verify_pack.py"
        result = subprocess.run([sys.executable, str(verifier)], capture_output=True, text=True, check=False)
        assert result.returncode == 0, f"subpack verification failed: {relative}: {(result.stdout + result.stderr).strip()}"


def check_year_map() -> None:
    map_text = (ROOT / "year-sequence.md").read_text(encoding="utf-8")
    weeks = [int(n) for n in re.findall(r"^\|\s*(\d+)\s*\|", map_text, re.MULTILINE)]
    assert weeks == list(range(1, 41)), f"Year 8 map is not Weeks 1-40: {weeks}"
    assert "Weeks 5–40" in map_text and "plan only" in map_text.lower()


def check_numbers_and_simulations() -> None:
    """Check the consequential prices and reproducible sample claims."""
    counts = (0, 2, 4, 6, 8)
    assert [18 + 7 * n for n in counts] == [18, 32, 46, 60, 74]
    assert [10 * n for n in counts] == [0, 20, 40, 60, 80]
    assert 18 + 7 * 6 == 10 * 6 == 60
    assert (18 + 7 * 8 + 9, 10 * 8) == (83, 80)
    assert 12 + 5 * 4 == 8 * 4 == 32
    travel = [0] * 40 + [1] * 30 + [2] * 20 + [3] * 10
    for seed, expected in ((1, [8, 7, 3, 2]), (67, [10, 5, 4, 1])):
        selected = random.Random(seed).sample(range(100), 20)
        assert [sum(travel[i] == category for i in selected) for category in range(4)] == expected
    for seed, expected in ((5, 11), (4, 14)):
        selected = random.Random(seed).sample(range(100), 20)
        assert sum(i < 60 for i in selected) == expected
    mean_errors = []
    for sample_size in (10, 40):
        errors = []
        for seed in range(1000):
            selected = random.Random(seed).sample(range(100), sample_size)
            bicycle_share = sum(travel[i] == 2 for i in selected) / sample_size
            errors.append(abs(bicycle_share - 0.2))
        mean_errors.append(round(100 * sum(errors) / len(errors), 2))
    assert mean_errors == [8.95, 3.92], mean_errors


def main() -> None:
    write = len(sys.argv) == 2 and sys.argv[1] == "--write-manifest"
    assert len(sys.argv) == 1 or write, __doc__
    expected = expected_manifest()
    if write:
        MANIFEST.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
    actual = json.loads(MANIFEST.read_text(encoding="utf-8"))
    assert actual == expected, "manifest hash/content drift; review edits, then use --write-manifest"
    stem = ROOT / "term-1/weeks-01-02"
    check_lesson_minutes(stem / "english/LESSONS.md", list(range(1, 11)), 25)
    check_lesson_minutes(stem / "mathematics/LESSONS.md", list(range(1, 11)), 25)
    for week in (1, 2):
        check_lesson_minutes(stem / f"integrated/week-0{week}.md", list(range(1, 6)), 35)
    check_year_map()
    check_numbers_and_simulations()
    check_later_subpacks()
    links = check_links()
    svgs, pdfs = check_print()
    print(f"Year 8 pack verified: 40×25-minute English/maths lessons; 20×35-minute integrated sessions; 50 optional five-area blocks; {len(actual['codes'])} exact Year 8/band codes; {len(actual['files'])} hashed files; {links} links; {svgs} opening SVGs/{pdfs} PDFs plus verified later aids; price and seeded-sample checks.")


if __name__ == "__main__":
    main()
