#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Check the bounded Queensland Year 12 opening pack and its SHA-256 receipt.

Run: python3 products/curriculum-studio/content/year-12/verify_pack.py
After human review of an edit: run with --write-manifest, then run normally.
This checks local structure, arithmetic and links. It does not establish QCAA
approval, classroom effectiveness, live syllabus currency or translation quality.
"""

from __future__ import annotations

import hashlib
import math
import re
import sys
from pathlib import Path
from urllib.parse import unquote
from xml.etree import ElementTree

ROOT = Path(__file__).resolve().parent
MANIFEST = ROOT / "manifest.sha256"
LESSONS = {
    ROOT / "english/term-1/weeks-01-02/LESSONS.md": (10, 25),
    ROOT / "mathematics/term-1/weeks-01-02/LESSONS.md": (10, 25),
    ROOT / "term-1/integrated/week-01.md": (5, 35),
    ROOT / "term-1/integrated/week-02.md": (5, 35),
}


def files() -> list[Path]:
    return sorted(
        p for p in ROOT.rglob("*")
        if p.is_file() and p != MANIFEST and "__pycache__" not in p.parts
    )


def check_lessons() -> int:
    for path, (n, total) in LESSONS.items():
        assert path.is_file(), f"missing authored sequence: {path}"
        content = path.read_text(encoding="utf-8")
        headings = list(re.finditer(r"^## Day (\d+)\b", content, re.MULTILINE))
        expected_days = list(range(6, 11)) if path.name == "week-02.md" else list(range(1, n + 1))
        assert [int(h.group(1)) for h in headings] == expected_days, path
        for i, h in enumerate(headings):
            end = headings[i + 1].start() if i + 1 < n else len(content)
            block = content[h.end():end]
            steps = re.findall(r"^([1-6])\. \*\*[^*]+ · (\d+) min\.\*\*", block, re.MULTILINE)
            assert [int(s[0]) for s in steps] == list(range(1, 7)), (path, i + 1, steps)
            assert sum(int(s[1]) for s in steps) == total, (path, i + 1, steps)
            assert "**Success:**" in block and "**Exit · 2 min.**" in block, (path, i + 1)
    return sum(n for n, _ in LESSONS.values())


def check_links() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        content = path.read_text(encoding="utf-8")
        for target in re.findall(r"\[[^\]]+\]\(([^)]+)\)", content):
            if re.match(r"^[a-z]+://", target) or target.startswith("mailto:"):
                continue
            local, _, anchor = unquote(target).partition("#")
            resolved = (path.parent / local).resolve() if local else path.resolve()
            assert resolved.is_file() and ROOT in resolved.parents, f"broken link {path}: {target}"
            if anchor and resolved.suffix.lower() == ".md":
                headings = re.findall(r"^#{1,6} +(.+?) *$", resolved.read_text(encoding="utf-8"), re.MULTILINE)
                slugs = {re.sub(r"[^\w -]", "", h.lower()).replace(" ", "-") for h in headings}
                assert anchor in slugs, f"broken anchor {path}: {target}"
            count += 1
    return count


def check_visuals() -> int:
    folder = ROOT / "term-1/print"
    svgs = sorted(folder.glob("*.svg"))
    pngs = sorted(folder.glob("*.png"))
    assert len(svgs) == len(pngs) == 3, (svgs, pngs)
    alternatives = (folder / "text-alternatives.md").read_text(encoding="utf-8")
    for svg in svgs:
        root = ElementTree.parse(svg).getroot()
        ns = "{http://www.w3.org/2000/svg}"
        assert root.get("width") == "297mm" and root.get("height") == "210mm", svg
        assert root.get("viewBox") == "0 0 1122 794" and root.get("role") == "img", svg
        assert root.get("aria-labelledby") == "title desc", svg
        assert root.find(f"{ns}title") is not None and root.find(f"{ns}desc") is not None, svg
        assert svg.name in alternatives and "CC BY 4.0" in svg.read_text(encoding="utf-8"), svg
        png = svg.with_suffix(".png")
        assert png.read_bytes()[:8] == b"\x89PNG\r\n\x1a\n", png
    for value in ("12/20 = 60%", "6/20 = 30%", "(1,8)", "(6,16)", "last bus"):
        assert value in alternatives, f"visual text alternative missing {value}"
    return len(svgs)


def corr(xs: list[int], ys: list[int]) -> tuple[float, float]:
    xbar = sum(xs) / len(xs)
    ybar = sum(ys) / len(ys)
    cross = sum((x - xbar) * (y - ybar) for x, y in zip(xs, ys, strict=True))
    xx = sum((x - xbar) ** 2 for x in xs)
    yy = sum((y - ybar) ** 2 for y in ys)
    value = cross / math.sqrt(xx * yy)
    return value, value**2


def check_calculations() -> None:
    assert (12 + 8, 6 + 14, 12 + 6, 8 + 14) == (20, 20, 18, 22)
    assert (12 / 20, 6 / 20, (12 - 6) / 20) == (0.6, 0.3, 0.3)
    assert (9 + 6, 4 + 11, 9 + 4, 6 + 11) == (15, 15, 13, 17)
    assert math.isclose((9 / 15 - 4 / 15) * 100, 100 / 3)
    for ys, expected_r, expected_r2 in (
        ([8, 10, 9, 14, 13, 16], 0.9189132409545225, 0.8444015444015444),
        ([5, 7, 10, 9, 13, 12], 0.9230930832457442, 0.8521008403361345),
    ):
        actual_r, actual_r2 = corr([1, 2, 3, 4, 5, 6], ys)
        assert math.isclose(actual_r, expected_r, abs_tol=1e-12)
        assert math.isclose(actual_r2, expected_r2, abs_tol=1e-12)
    assert 8 / 10 == 0.8


def check_scope() -> None:
    intro = (ROOT / "README.md").read_text(encoding="utf-8")
    map_text = (ROOT / "year-sequence.md").read_text(encoding="utf-8")
    ledger = (ROOT / "source-and-qa.md").read_text(encoding="utf-8")
    assert "Queensland General" in intro and "no national acara year 12" in intro.lower(), intro[:500]
    slots = re.findall(r"^\| (\d+) \|", map_text, re.MULTILINE)
    assert slots == [str(n) for n in range(1, 41)], f"missing planning slots: {slots}"
    assert "Slot 3–40 is plan only" in map_text
    planning_rows = [line for line in map_text.splitlines() if re.match(r"^\| (?:[3-9]|[1-3][0-9]|40) \|", line)]
    assert len(planning_rows) == 38 and all(line.endswith("| Plan only |") for line in planning_rows)
    for ref in ("ENG-U3-T1", "ENG-TEXTS", "GM-U3-T1-CAT", "GM-U3-T1-NUM", "QCAA-COURSE"):
        assert ref in ledger, ref
    for name in ("student-materials.md", "student-assessment.md", "teacher-key.md"):
        for subject in ("english", "mathematics"):
            assert (ROOT / subject / "term-1/weeks-01-02" / name).is_file()
    assert (ROOT / "term-1/integrated/student-materials.md").is_file()
    assert (ROOT / "term-1/integrated/teacher-key.md").is_file()
    all_md = "\n".join(p.read_text(encoding="utf-8") for p in ROOT.rglob("*.md"))
    assert not re.search(r"\bAC9[A-Z0-9]{5,}\b", all_md), "national F–10 code in senior pack"


def check_clean_files() -> None:
    for path in files():
        assert path.suffix in {".md", ".py", ".svg", ".png"}, path
        if path.suffix != ".png":
            data = path.read_text(encoding="utf-8")
            assert not any(line.rstrip(" \t") != line for line in data.splitlines()), f"trailing whitespace: {path}"


def expected_manifest() -> str:
    return "".join(
        f"{hashlib.sha256(p.read_bytes()).hexdigest()}  {p.relative_to(ROOT).as_posix()}\n"
        for p in files()
    )


def main() -> None:
    n = check_lessons()
    links = check_links()
    aids = check_visuals()
    check_calculations()
    check_scope()
    check_clean_files()
    receipt = expected_manifest()
    if "--write-manifest" in sys.argv[1:]:
        MANIFEST.write_text(receipt, encoding="utf-8")
    else:
        assert MANIFEST.read_text(encoding="utf-8") == receipt, "manifest drift; review before --write-manifest"
    print(f"PASS: {n} timed sessions (20 subject + 10 optional integrated); {links} local links; {aids} accessible SVG aids; arithmetic, scope and {len(files())} hashes")


if __name__ == "__main__":
    main()
