# SPDX-License-Identifier: Apache-2.0
"""Check the authored Year 5 opening four weeks and refresh the receipt.

Run from anywhere: python3 products/curriculum-studio/content/year-5/verify_pack.py
To update after reviewing edits: add --write-manifest. No network or packages needed.
"""

from __future__ import annotations

import hashlib
import json
import re
import subprocess
import sys
from pathlib import Path
from urllib.parse import unquote
from xml.etree import ElementTree

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[1]
SOURCE = STUDIO / "data/frameworks/acara-v9.json"
MANIFEST = ROOT / "manifest.json"
WORKBOOK_HASH = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
SOURCE_URL = "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx"
TERMS_URL = "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use"
REQUIRED_CODES = {
    "AC9E5LA02", "AC9E5LA04", "AC9E5LA05", "AC9E5LA06", "AC9E5LY02", "AC9E5LY05", "AC9E5LY06", "AC9E5LY07",
    "AC9M5N01", "AC9M5N02", "AC9M5N09",
    "AC9S5U03", "AC9S5I01", "AC9S5I02", "AC9S5I03", "AC9S5I04", "AC9S5I05", "AC9S5I06",
    "AC9HS5K08", "AC9HS5S01", "AC9HS5S03", "AC9HS5S05", "AC9HS5S06",
}


def digest(data: bytes) -> str:
    return hashlib.sha256(data).hexdigest()


def assets() -> list[Path]:
    return sorted(
        p for p in ROOT.rglob("*")
        if p.is_file() and p != MANIFEST and "__pycache__" not in p.parts
    )


def rights(path: Path) -> str:
    if path.name == "FONT-RIGHTS.md":
        return "Third-party DejaVu font notice under Bitstream Vera licence; not SubjectNest CC BY"
    if path.suffix == ".pdf":
        return "Original SubjectNest design CC BY 4.0; embedded DejaVu font subsets under separate FONT-RIGHTS.md licence"
    if path.suffix == ".py":
        return "Apache-2.0 verification code"
    return "CC BY 4.0; original SubjectNest material"


def curriculum_records() -> dict[str, dict]:
    data = json.loads(SOURCE.read_text(encoding="utf-8"))
    assert data["source_sha256"] == WORKBOOK_HASH, "ACARA workbook hash drift"
    assert data["source_url"] == SOURCE_URL, "ACARA workbook URL drift"
    return {
        r["code"]: r
        for r in data["records"]
        if r["record_type"] == "content_description" and r["attributes"].get("level") in {"Year 5", "Years 5 and 6"}
    }


def code_refs() -> set[str]:
    refs: set[str] = set()
    for path in ROOT.rglob("*.md"):
        refs.update(re.findall(r"\bAC9[A-Z0-9]+\b", path.read_text(encoding="utf-8")))
    return refs


def expected_manifest() -> dict:
    source = curriculum_records()
    refs = code_refs()
    assert REQUIRED_CODES <= refs, f"required code not present in source notes: {sorted(REQUIRED_CODES - refs)}"
    assert refs <= source.keys(), f"wrong-level or absent Year 5/Years 5 and 6 codes: {sorted(refs - source.keys())}"
    return {
        "pack": "SubjectNest Year 5 opening four weeks",
        "version": "0.1.0-draft",
        "status": "Weeks 1-4 English and mathematics authored; Weeks 1-4 integrated project blocks authored; Weeks 3-4 five-area optional supplementary menu authored; full subject allocations, later weeks and local human review pending",
        "original_author": "NeuroForgeIO Pty Ltd",
        "curriculum_source": {
            "publisher": "ACARA",
            "framework": "Australian Curriculum Version 9.0",
            "source_url": SOURCE_URL,
            "retrieved_at": "2026-09-29",
            "workbook_sha256": WORKBOOK_HASH,
            "terms_url": TERMS_URL,
            "not_endorsed_by_acara": True,
        },
        "codes": {
            code: {
                "level": source[code]["attributes"]["level"],
                "learning_area": source[code]["attributes"].get("learning_area", ""),
                "description_sha256": digest(source[code]["plain_text"].encode("utf-8")),
            }
            for code in sorted(refs)
        },
        "files": [
            {
                "path": path.relative_to(ROOT).as_posix(),
                "sha256": digest(path.read_bytes()),
                "rights": rights(path),
            }
            for path in assets()
        ],
        "rights_note": "Original lessons, fictional data, original WAV cue, print designs and visuals CC BY 4.0 with credit, source, licence and change notice. Curriculum remains under ACARA terms. Embedded PDF DejaVu font subsets retain separate font licences documented in pack font notices. No third-party audio, image or translated assets included.",
    }


def check_lesson_minutes(path: Path, days: list[int], minutes: int) -> None:
    content = path.read_text(encoding="utf-8")
    headers = list(re.finditer(r"^## Day (\d+)\b", content, re.MULTILINE))
    actual = [int(h.group(1)) for h in headers]
    assert actual == days, f"{path}: day order {actual}, expected {days}"
    for i, header in enumerate(headers):
        end = headers[i + 1].start() if i + 1 < len(headers) else len(content)
        block = content[header.end():end]
        steps = [int(x) for x in re.findall(r"^\d+\. \*\*[^\n*]*? · (\d+) min\.\*\*", block, re.MULTILINE)]
        assert len(steps) == 6 and sum(steps) == minutes, (path, actual[i], steps)
        assert "**Success:**" in block or minutes == 35, (path, actual[i], "missing success")


def check_links() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        for target in re.findall(r"\[[^\]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if re.match(r"^[a-z]+://", target) or target.startswith("mailto:"):
                continue
            local, _, anchor = unquote(target).partition("#")
            resolved = (path.parent / local).resolve() if local else path.resolve()
            assert resolved.is_file() and ROOT in resolved.parents, f"broken or out-of-pack link: {path}: {target}"
            if anchor and resolved.suffix.lower() == ".md":
                text = resolved.read_text(encoding="utf-8")
                headings = re.findall(r"^#{1,6} +(.+?) *$", text, re.MULTILINE)
                slug_set = {
                    re.sub(r"[^\w -]", "", re.sub(r"<[^>]+>", "", h).lower()).replace(" ", "-")
                    for h in headings
                }
                assert anchor in slug_set, f"broken anchor: {path}: {target}"
            count += 1
    return count


def check_print() -> tuple[int, int]:
    svgs = list(ROOT.rglob("*.svg"))
    pdfs = list(ROOT.rglob("*.pdf"))
    assert len(svgs) == len(pdfs) == 13, "expected thirteen paired editable/print aids"
    font_notice = ROOT / "term-1/weeks-01-02/print/FONT-RIGHTS.md"
    assert "Bitstream Vera" in font_notice.read_text(encoding="utf-8"), "missing embedded-font rights notice"
    for svg in svgs:
        element = ElementTree.parse(svg).getroot()
        ns = "{http://www.w3.org/2000/svg}"
        assert element.get("viewBox") and element.get("role") == "img", svg
        assert element.get("aria-labelledby") == "title desc", svg
        assert element.find(f"{ns}title") is not None and element.find(f"{ns}desc") is not None, svg
        if "CC BY 4.0" not in svg.read_text(encoding="utf-8"):
            ledger = svg.parent.parent / "SOURCE-AND-RIGHTS.md"
            assert ledger.is_file() and "CC BY 4.0" in ledger.read_text(encoding="utf-8"), svg
        assert svg.with_suffix(".pdf").read_bytes().startswith(b"%PDF"), svg
    alternatives = (ROOT / "term-1/weeks-01-02/print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    for phrase in ("1.230", "1.260", "1×36", "What this cannot settle"):
        assert phrase in alternatives, f"missing text equivalent: {phrase}"
    return len(svgs), len(pdfs)


def main() -> None:
    write = len(sys.argv) == 2 and sys.argv[1] == "--write-manifest"
    assert len(sys.argv) == 1 or write, __doc__
    expected = expected_manifest()
    if write:
        MANIFEST.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
    actual = json.loads(MANIFEST.read_text(encoding="utf-8"))
    assert actual == expected, "manifest hash/content drift; review edits, then use --write-manifest"
    stem = ROOT / "term-1/weeks-01-02"
    check_lesson_minutes(stem / "english/LESSONS.md", list(range(1, 11)), 25)
    check_lesson_minutes(stem / "mathematics/LESSONS.md", list(range(1, 11)), 25)
    for week in (1, 2):
        check_lesson_minutes(stem / f"integrated/week-0{week}.md", list(range(1, 6)), 35)
    for subject in ("english", "mathematics"):
        verifier = ROOT / subject / "term-1/weeks-03-04/verify_pack.py"
        assert verifier.is_file(), f"missing Weeks 3–4 {subject} verifier"
        subprocess.run([sys.executable, str(verifier)], check=True, capture_output=True, text=True)
    for area in ("integrated", "supplementary"):
        verifier = ROOT / area / "term-1/weeks-03-04/verify_pack.py"
        assert verifier.is_file(), f"missing Weeks 3–4 {area} verifier"
        subprocess.run([sys.executable, str(verifier)], check=True, capture_output=True, text=True)
    links = check_links()
    svgs, pdfs = check_print()
    print(f"Year 5 pack verified: 40×25-minute daily English/maths lessons; 20×35-minute integrated sessions; 50 optional 12-minute supplementary blocks; {len(actual['codes'])} exact Year 5/Years 5 and 6 codes; {len(actual['files'])} hashed files; {links} links; {svgs} SVGs/{pdfs} PDFs.")


if __name__ == "__main__":
    main()
