#!/usr/bin/env python3
"""Read-only validation for the SubjectNest Year 9 opening pack.

Copyright 2026 NeuroForgeIO Pty Ltd
SPDX-License-Identifier: Apache-2.0

Run from any directory:
python3 products/curriculum-studio/content/year-9/verify_pack.py
"""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import struct
import subprocess
import sys
import xml.etree.ElementTree as ET
from pathlib import Path

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parent.parent
CANONICAL = STUDIO / "data/frameworks/acara-v9.json"
WORKBOOK = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
WORKBOOK_SHA256 = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODE_RE = re.compile(r"AC9[A-Z0-9]+")
LINK_RE = re.compile(r"\[[^]]+\]\(([^)]+)\)")
DAY_RE = re.compile(r"^## Day (\d+)\b[^\n]*$", re.MULTILINE)
MOVE_RE = re.compile(r"^\d+\. \*\*[^*]*?(\d+):\*\*", re.MULTILINE)
LATER_PACKS = (
    ROOT / "english/term-1/weeks-03-04",
    ROOT / "mathematics/term-1/weeks-03-04",
    ROOT / "integrated/term-1/weeks-03-04",
    ROOT / "supplementary/term-1/weeks-03-04",
)


def in_later_pack(path: Path) -> bool:
    return any(path.is_relative_to(pack) for pack in LATER_PACKS)


def digest(path: Path) -> str:
    return hashlib.sha256(path.read_bytes()).hexdigest()


def write_root_manifest() -> None:
    manifest = ROOT / "manifest.sha256"
    entries = sorted(path for path in ROOT.rglob("*") if path.is_file() and path != manifest)
    manifest.write_text(
        "".join(f"{digest(path)}  {path.relative_to(ROOT).as_posix()}\n" for path in entries),
        encoding="utf-8",
    )


def year9_codes(node: object) -> set[str]:
    codes: set[str] = set()

    def visit(value: object) -> None:
        if isinstance(value, dict):
            attrs = value.get("attributes", {})
            if (
                value.get("record_type") == "content_description"
                and isinstance(attrs, dict)
                and attrs.get("level") in {"Year 9", "Years 9 and 10"}
                and isinstance(value.get("code"), str)
            ):
                codes.add(value["code"])
            for inner in value.values():
                visit(inner)
        elif isinstance(value, list):
            for inner in value:
                visit(inner)

    visit(node)
    return codes


def heading_slugs(body: str) -> set[str]:
    result: set[str] = set()
    for title in re.findall(r"^#{1,6}\s+(.+)$", body, re.MULTILINE):
        plain = title.replace("`", "").lower()
        plain = "".join(ch for ch in plain if ch.isalnum() or ch in " -_")
        result.add(plain.replace(" ", "-"))
    return result


def check_markdown(valid_codes: set[str], errors: list[str]) -> int:
    used: set[str] = set()
    for path in ROOT.rglob("*.md"):
        body = path.read_text(encoding="utf-8")
        used.update(CODE_RE.findall(body))
        if not in_later_pack(path) and "CC BY 4.0" not in body:
            errors.append(f"original-material rights absent: {path.relative_to(ROOT)}")
        for target in LINK_RE.findall(body):
            if target.startswith(("https://", "http://")):
                continue
            filename, _, anchor = target.partition("#")
            linked = (path.parent / filename).resolve() if filename else path.resolve()
            if not linked.is_relative_to(ROOT):
                errors.append(f"link escapes pack: {path.relative_to(ROOT)} -> {target}")
            elif not linked.exists():
                errors.append(f"broken local link: {path.relative_to(ROOT)} -> {target}")
            elif anchor and linked.suffix == ".md" and anchor not in heading_slugs(linked.read_text(encoding="utf-8")):
                errors.append(f"broken heading anchor: {path.relative_to(ROOT)} -> {target}")
    for code in sorted(used - valid_codes):
        errors.append(f"not an exact Year 9 content-description code: {code}")
    return len(used)


def check_day_file(path: Path, count: int, minutes: int, errors: list[str]) -> None:
    if not path.is_file():
        errors.append(f"missing lessons: {path.relative_to(ROOT)}")
        return
    body = path.read_text(encoding="utf-8")
    matches = list(DAY_RE.finditer(body))
    numbers = [int(m.group(1)) for m in matches]
    if numbers != list(range(1, count + 1)):
        errors.append(f"day headings {path.relative_to(ROOT)}: {numbers}")
    for idx, match in enumerate(matches):
        end = matches[idx + 1].start() if idx + 1 < len(matches) else len(body)
        moves = [int(m) for m in MOVE_RE.findall(body[match.end() : end])][:5]
        if len(moves) != 5 or sum(moves) != minutes:
            errors.append(f"timing {path.relative_to(ROOT)} Day {match.group(1)}: {moves}")


def check_sequence(errors: list[str]) -> None:
    body = (ROOT / "year-sequence.md").read_text(encoding="utf-8")
    rows = re.findall(r"^\| (\d+) \|(.+)$", body, re.MULTILINE)
    weeks = [int(number) for number, _ in rows]
    if weeks != list(range(1, 41)):
        errors.append(f"planning weeks not 1–40 in order: {weeks}")
    for number, rest in rows:
        if int(number) >= 5 and "Plan only" not in rest:
            errors.append(f"Week {number} lacks plan-only label")
        if int(number) in {3, 4} and "Authored" not in rest:
            errors.append(f"Week {number} lacks authored pack links")
    for subject in ("mathematics", "english"):
        check_day_file(ROOT / subject / "term-1/weeks-01-02/lessons.md", 10, 25, errors)
    for name in ("week-01-civic-media-lab.md", "week-02-energy-systems-lab.md"):
        check_day_file(ROOT / "term-1" / name, 5, 35, errors)


def check_visuals(errors: list[str]) -> int:
    svgs = sorted(path for path in ROOT.rglob("*.svg") if not in_later_pack(path))
    for path in svgs:
        try:
            ET.parse(path)
        except ET.ParseError as exc:
            errors.append(f"invalid SVG {path.relative_to(ROOT)}: {exc}")
        body = path.read_text(encoding="utf-8")
        if "<title" not in body or "<desc" not in body:
            errors.append(f"missing SVG text alternative: {path.relative_to(ROOT)}")
        if re.search(r"@font-face|data:|<image\b|font-family=\"(?!sans-serif\")", body):
            errors.append(f"embedded/external image or font needs rights review: {path.relative_to(ROOT)}")
        if not path.with_suffix(".png").is_file():
            errors.append(f"missing PNG render: {path.relative_to(ROOT)}")
    for path in (path for path in ROOT.rglob("*.png") if not in_later_pack(path)):
        header = path.read_bytes()[:24]
        if len(header) != 24 or header[:8] != b"\x89PNG\r\n\x1a\n":
            errors.append(f"invalid PNG: {path.relative_to(ROOT)}")
            continue
        width, height = struct.unpack(">II", header[16:24])
        if width < 700 or height < 700:
            errors.append(f"render too small: {path.relative_to(ROOT)} {width}x{height}")
    if len(svgs) != 4:
        errors.append(f"expected four original SVG assets, found {len(svgs)}")
    return len(svgs)


def check_manifests(errors: list[str]) -> int:
    manifests = sorted(ROOT.rglob("manifest.sha256"))
    if len(manifests) != 4:
        errors.append(f"expected four manifests, found {len(manifests)}")
    for manifest in manifests:
        listed: set[Path] = set()
        for line in manifest.read_text(encoding="utf-8").splitlines():
            match = re.fullmatch(r"([0-9a-f]{64})  (.+)", line)
            if not match:
                errors.append(f"bad manifest row: {manifest.relative_to(ROOT)}: {line}")
                continue
            hash_value, name = match.groups()
            path = (manifest.parent / name).resolve()
            if not path.is_relative_to(ROOT) or path in listed or not path.is_file() or path == manifest:
                errors.append(f"duplicate/missing/self item: {manifest.relative_to(ROOT)} -> {name}")
                continue
            listed.add(path)
            if digest(path) != hash_value:
                errors.append(f"hash mismatch: {path.relative_to(ROOT)}")
        if manifest == ROOT / "manifest.sha256":
            expected = {p.resolve() for p in ROOT.rglob("*") if p.is_file() and p != manifest}
        else:
            expected = {p.resolve() for p in manifest.parent.iterdir() if p.is_file() and p != manifest}
        if listed != expected:
            missing = sorted(str(p.relative_to(ROOT)) for p in expected - listed)
            extra = sorted(str(p.relative_to(ROOT)) for p in listed - expected)
            errors.append(f"manifest inventory {manifest.relative_to(ROOT)} missing={missing} extra={extra}")
    return len(manifests)


def check_later_packs(errors: list[str]) -> None:
    for pack in LATER_PACKS:
        verifier = pack / "verify_pack.py"
        if not verifier.is_file():
            errors.append(f"missing later-pack verifier: {pack.relative_to(ROOT)}")
            continue
        result = subprocess.run(
            [sys.executable, str(verifier)],
            cwd=pack,
            capture_output=True,
            text=True,
            check=False,
        )
        if result.returncode:
            errors.append(f"later-pack verification failed: {pack.relative_to(ROOT)}: {result.stdout} {result.stderr}")


def main() -> int:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--write-manifest", action="store_true", help="refresh the root SHA-256 inventory before validating")
    args = parser.parse_args()
    if args.write_manifest:
        write_root_manifest()
    errors: list[str] = []
    if not CANONICAL.is_file() or not WORKBOOK.is_file():
        print("Pinned ACARA canonical import or workbook missing", file=sys.stderr)
        return 1
    if digest(WORKBOOK) != WORKBOOK_SHA256:
        errors.append("official pinned workbook SHA-256 changed")
    valid_codes = year9_codes(json.loads(CANONICAL.read_text(encoding="utf-8")))
    used = check_markdown(valid_codes, errors)
    check_sequence(errors)
    visual_count = check_visuals(errors)
    manifest_count = check_manifests(errors)
    check_later_packs(errors)
    if errors:
        for error in errors:
            print(f"FAIL {error}", file=sys.stderr)
        return 1
    print(
        f"PASS Year 9: {used} exact Year 9/band content-code tokens, 40 mapped weeks "
        f"(Weeks 1–4 authored, 5–40 plan only), 40 English/mathematics lessons, "
        f"20 integrated sessions, 50 optional five-area blocks; {visual_count} opening SVG/PNG aids, "
        f"{manifest_count} opening/root SHA manifests and four verified later packs"
    )
    return 0


if __name__ == "__main__":
    raise SystemExit(main())
