#!/usr/bin/env python3
"""Verify the SubjectNest Year 6 authored opening sequence from pinned sources.

Copyright 2026 NeuroForgeIO Pty Ltd
SPDX-License-Identifier: Apache-2.0

Run from any directory: python3 products/curriculum-studio/content/year-6/verify_pack.py
The verifier reads the canonical ACARA import and workbook; it writes nothing.
"""

from __future__ import annotations

import hashlib
import json
import re
import struct
import subprocess
import sys
import xml.etree.ElementTree as ET
from pathlib import Path

ROOT = Path(__file__).resolve().parent
CURRICULUM_STUDIO = ROOT.parent.parent
CANONICAL = CURRICULUM_STUDIO / "data/frameworks/acara-v9.json"
WORKBOOK = (
    CURRICULUM_STUDIO
    / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
)
WORKBOOK_SHA256 = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODE_RE = re.compile(r"AC9[A-Z0-9]+")
LINK_RE = re.compile(r"\[[^]]+\]\(([^)]+)\)")
DAY_RE = re.compile(r"^## Day (\d+)\b[^\n]*$", re.MULTILINE)
MOVE_RE = re.compile(r"^\d+\. \*\*[^*]*?(\d+):\*\*", re.MULTILINE)


def sha256(path: Path) -> str:
    return hashlib.sha256(path.read_bytes()).hexdigest()


def headings_slug(text: str) -> set[str]:
    result: set[str] = set()
    for heading in re.findall(r"^#{1,6}\s+(.+)$", text, re.MULTILINE):
        plain = heading.replace("`", "").lower()
        plain = "".join(char for char in plain if char.isalnum() or char in " -_")
        result.add(plain.replace(" ", "-"))
    return result


def canonical_year6_codes(data: object) -> set[str]:
    result: set[str] = set()

    def visit(node: object) -> None:
        if isinstance(node, dict):
            if (
                node.get("record_type") == "content_description"
                and isinstance(node.get("attributes"), dict)
                and node["attributes"].get("level") in {"Year 6", "Years 5 and 6"}
                and isinstance(node.get("code"), str)
            ):
                result.add(node["code"])
            for value in node.values():
                visit(value)
        elif isinstance(node, list):
            for value in node:
                visit(value)

    visit(data)
    return result


def verify_markdown(codes: set[str], errors: list[str]) -> int:
    used: set[str] = set()
    for path in ROOT.rglob("*.md"):
        body = path.read_text(encoding="utf-8")
        used.update(CODE_RE.findall(body))
        # Later subpacks carry their rights at pack level and verify locally.
        if "weeks-03-04" not in path.parts and "CC BY 4.0" not in body:
            errors.append(f"missing original-material rights: {path.relative_to(ROOT)}")
        for target in LINK_RE.findall(body):
            if target.startswith(("https://", "http://")):
                continue
            filename, _, anchor = target.partition("#")
            linked = (path.parent / filename).resolve() if filename else path.resolve()
            if not linked.exists():
                errors.append(f"broken local link: {path.relative_to(ROOT)} -> {target}")
            elif anchor and linked.suffix == ".md" and anchor not in headings_slug(linked.read_text(encoding="utf-8")):
                errors.append(f"broken heading anchor: {path.relative_to(ROOT)} -> {target}")
    for code in sorted(used - codes):
        errors.append(f"not a Year 6/Years 5 and 6 content-description code: {code}")
    return len(used)


def verify_sequence(errors: list[str]) -> None:
    planning = (ROOT / "year-sequence.md").read_text(encoding="utf-8")
    weeks = [int(n) for n in re.findall(r"^\| (\d+) \|", planning, re.MULTILINE)]
    if weeks != list(range(1, 41)):
        errors.append(f"planning weeks are not 1–40 in order: {weeks}")
    for subject in ("mathematics", "english"):
        path = ROOT / subject / "term-1/weeks-01-02/lessons.md"
        verify_days(path, 10, 25, errors)
    for name in ("week-01-choice-lab.md", "week-02-sun-earth-model.md"):
        verify_days(ROOT / "term-1" / name, 5, 35, errors)


def verify_later_subpacks(errors: list[str]) -> None:
    for relative in (
        "english/term-1/weeks-03-04",
        "mathematics/term-1/weeks-03-04",
        "integrated/term-1/weeks-03-04",
        "supplementary/term-1/weeks-03-04",
    ):
        verifier = ROOT / relative / "verify_pack.py"
        result = subprocess.run([sys.executable, str(verifier)], capture_output=True, text=True, check=False)
        if result.returncode:
            errors.append(f"subpack verification failed: {relative}: {(result.stdout + result.stderr).strip()}")


def verify_days(path: Path, expected_days: int, expected_minutes: int, errors: list[str]) -> None:
    body = path.read_text(encoding="utf-8")
    matches = list(DAY_RE.finditer(body))
    numbers = [int(match.group(1)) for match in matches]
    if numbers != list(range(1, expected_days + 1)):
        errors.append(f"day headings out of order in {path.relative_to(ROOT)}: {numbers}")
    for index, match in enumerate(matches):
        end = matches[index + 1].start() if index + 1 < len(matches) else len(body)
        moves = [int(n) for n in MOVE_RE.findall(body[match.end() : end])]
        # End material after the final lesson may contain numbered prose; only
        # the first five move rows define the day.
        moves = moves[:5]
        if len(moves) != 5 or sum(moves) != expected_minutes:
            errors.append(
                f"timing in {path.relative_to(ROOT)} Day {match.group(1)}: {moves}"
            )


def verify_visuals(errors: list[str]) -> int:
    svg_paths = list(ROOT.rglob("*.svg"))
    for path in svg_paths:
        try:
            ET.parse(path)
        except ET.ParseError as exc:
            errors.append(f"invalid SVG {path.relative_to(ROOT)}: {exc}")
        if not (path.with_suffix(".png").exists() or path.with_suffix(".pdf").exists()):
            errors.append(f"missing rendered PNG/PDF for {path.relative_to(ROOT)}")
        body = path.read_text(encoding="utf-8")
        if "<title" not in body or "<desc" not in body:
            errors.append(f"missing SVG text alternative: {path.relative_to(ROOT)}")
    for path in ROOT.rglob("*.png"):
        header = path.read_bytes()[:24]
        if len(header) != 24 or header[:8] != b"\x89PNG\r\n\x1a\n":
            errors.append(f"invalid PNG signature: {path.relative_to(ROOT)}")
            continue
        width, height = struct.unpack(">II", header[16:24])
        if width < 700 or height < 700:
            errors.append(f"PNG preview too small: {path.relative_to(ROOT)} {width}×{height}")
    return len(svg_paths)


def verify_manifests(errors: list[str]) -> int:
    manifests = sorted(ROOT.rglob("manifest.sha256"))
    if len(manifests) != 4:
        errors.append(f"expected root, maths, English and integrated manifests; found {len(manifests)}")
    for manifest in manifests:
        listed: set[Path] = set()
        for line in manifest.read_text(encoding="utf-8").splitlines():
            match = re.fullmatch(r"([0-9a-f]{64})  (.+)", line)
            if not match:
                errors.append(f"invalid manifest row in {manifest.relative_to(ROOT)}: {line}")
                continue
            digest, name = match.groups()
            path = (manifest.parent / name).resolve()
            if path in listed or not path.is_file() or path == manifest:
                errors.append(f"duplicate/missing/self manifest item: {name}")
                continue
            listed.add(path)
            if sha256(path) != digest:
                errors.append(f"hash mismatch: {path.relative_to(ROOT)}")
        if manifest == ROOT / "manifest.sha256":
            expected = {p.resolve() for p in ROOT.rglob("*") if p.is_file() and p != manifest and "__pycache__" not in p.parts}
        else:
            expected = {p.resolve() for p in manifest.parent.iterdir() if p.is_file() and p != manifest}
        if listed != expected:
            missing = sorted(str(p.relative_to(ROOT)) for p in expected - listed)
            extra = sorted(str(p.relative_to(ROOT)) for p in listed - expected)
            errors.append(f"manifest inventory {manifest.relative_to(ROOT)} missing={missing} extra={extra}")
    return len(manifests)


def main() -> int:
    errors: list[str] = []
    if not CANONICAL.is_file() or not WORKBOOK.is_file():
        print("Pinned ACARA canonical import or workbook is missing", file=sys.stderr)
        return 1
    if sha256(WORKBOOK) != WORKBOOK_SHA256:
        errors.append("official pinned workbook SHA-256 changed")
    codes = canonical_year6_codes(json.loads(CANONICAL.read_text(encoding="utf-8")))
    used_count = verify_markdown(codes, errors)
    verify_sequence(errors)
    verify_later_subpacks(errors)
    visual_count = verify_visuals(errors)
    manifest_count = verify_manifests(errors)
    if errors:
        for error in errors:
            print(f"FAIL {error}", file=sys.stderr)
        return 1
    print(
        f"PASS Year 6: {used_count} exact Year 6/band codes, 40 planning weeks, "
        f"40 English/maths lessons and 20 integrated sessions through Week 4, "
        f"50 optional five-area blocks in Weeks 3–4, {visual_count} SVG aids, "
        f"{manifest_count} opening/root hash manifests and 4 verified later subpacks"
    )
    return 0


if __name__ == "__main__":
    raise SystemExit(main())
