# SPDX-License-Identifier: Apache-2.0
"""Verify Year 5 decision-desk source, arithmetic, teaching and file receipt."""

from __future__ import annotations

import hashlib
import json
import re
import subprocess
import sys
from pathlib import Path
from urllib.parse import unquote
from xml.etree import ElementTree

ROOT = Path(__file__).resolve().parent
STUDIO = next(parent for parent in ROOT.parents if parent.name == "curriculum-studio")
WORKBOOK = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
SOURCE = STUDIO / "data/frameworks/acara-v9.json"
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
MANIFEST = ROOT / "MANIFEST.sha256"
CODES = {
    "AC9E5LA02", "AC9E5LA04", "AC9E5LY02", "AC9E5LY05", "AC9E5LY06",
    "AC9M5N06", "AC9M5N08", "AC9M5N09", "AC9M5A01", "AC9M5A02",
}


def digest(path: Path) -> str:
    return hashlib.sha256(path.read_bytes()).hexdigest()


def check_source() -> None:
    data = json.loads(SOURCE.read_text(encoding="utf-8"))
    assert digest(WORKBOOK) == SOURCE_SHA == data["source_sha256"]
    records = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description"}
    crosswalk = (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8")
    matched: set[str] = set()
    for line in crosswalk.splitlines():
        match = re.fullmatch(r"\| (AC9(?:E5|M5)[A-Z0-9]+) \| (\d+) \| Year 5 \| (.+?) \| (.+?) \|", line)
        if not match:
            continue
        code, row, wording, note = match.groups()
        assert code in CODES and code not in matched and note.strip()
        record = records[code]
        assert record["attributes"]["level"] == "Year 5"
        assert record["source_row"] == int(row)
        assert " ".join(record["plain_text"].split()) == wording
        matched.add(code)
    assert matched == CODES, (matched, CODES)


def check_days() -> None:
    teacher = (ROOT / "TEACHER-SEQUENCE.md").read_text(encoding="utf-8")
    headers = list(re.finditer(r"^### Day (\d+)\b", teacher, re.MULTILINE))
    assert [int(m.group(1)) for m in headers] == list(range(11, 21))
    for index, header in enumerate(headers):
        body = teacher[header.end(): headers[index + 1].start() if index + 1 < len(headers) else len(teacher)]
        minutes = [int(n) for n in re.findall(r"^\d+\. \*\*[^\n*]+ · (\d+) min\.\*\*", body, re.MULTILINE)]
        assert minutes == [5, 7, 10, 9, 4], (header.group(1), minutes)
        assert "**Aim:**" in body and "**Prepare:**" in body
    learner = (ROOT / "LEARNER-CARDS.md").read_text(encoding="utf-8")
    assert [int(n) for n in re.findall(r"^## Day (\d+)\b", learner, re.MULTILINE)] == list(range(11, 21))
    assert "TEACHER-KEY" not in learner and "**828**" not in learner and "**888**" not in learner
    assert "after your subject checks" in learner.lower()
    assert "after the separate independent subject checks" in teacher


def check_arithmetic() -> None:
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    cases = [(46, 18, 828, 72), (24, 37, 888, 12), (29, 32, 928, 28), (42, 21, 882, 18), (28, 32, 896, 4), (33, 26, 858, 42), (22, 41, 902, 2), (21, 41, 861, 39)]
    for groups, cards, product, gap in cases:
        assert groups * cards == product and abs(900 - product) == gap
        assert f"{groups}×{cards} = **{product} cards**" in key, (groups, cards)


def check_aid() -> None:
    svg = ROOT / "print/decision-mat.svg"
    pdf = svg.with_suffix(".pdf")
    tree = ElementTree.parse(svg).getroot()
    namespace = "{http://www.w3.org/2000/svg}"
    assert tree.get("width") == "210mm" and tree.get("height") == "297mm"
    assert tree.get("viewBox") == "0 0 210 297" and tree.get("role") == "img"
    assert tree.get("aria-labelledby") == "title desc"
    assert tree.find(f"{namespace}title") is not None and tree.find(f"{namespace}desc") is not None
    info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
    content = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
    assert "(A4)" in info and "Pages:           1" in info
    assert all(s in content for s in ("FAIR DECISION DESK", "EXACT GROUPS", "COMPARE WITH 900", "CC BY 4.0"))
    assert "Bitstream Vera" in (ROOT / "print/FONT-RIGHTS.md").read_text(encoding="utf-8")
    alt = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    assert all(s in alt for s in ("six large boxes", "tactile", "not asserted to be a tagged accessible PDF"))


def check_links() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        for url in re.findall(r"\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if re.match(r"^[a-z]+://", url):
                continue
            local, _, anchor = unquote(url).partition("#")
            target = (path.parent / local).resolve() if local else path.resolve()
            assert target.is_file(), (path, url)
            if anchor and target.suffix == ".md":
                headings = re.findall(r"^#{1,6} +(.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)
                slugs = {re.sub(r"[^\w -]", "", h.lower()).replace(" ", "-") for h in headings}
                assert anchor in slugs, (path, url)
            count += 1
    return count


def check_manifest(write: bool) -> int:
    files = sorted(p for p in ROOT.rglob("*") if p.is_file() and p != MANIFEST and "__pycache__" not in p.parts)
    expected = "".join(f"{digest(p)}  {p.relative_to(ROOT).as_posix()}\n" for p in files)
    if write:
        MANIFEST.write_text(expected, encoding="utf-8")
    assert MANIFEST.read_text(encoding="utf-8") == expected, "hash manifest drift"
    return len(files)


def main() -> None:
    write = sys.argv[1:] == ["--write-manifest"]
    assert not sys.argv[1:] or write
    check_source()
    check_days()
    check_arithmetic()
    check_aid()
    links = check_links()
    files = check_manifest(write)
    print(f"PASS Year 5 integrated Weeks 3–4: ten 35-minute blocks, 10 clean learner cards, 10 exact Year 5 codes, 8 recomputed plan cases, A4 aid/text route, {links} local links, {files} hashes")


if __name__ == "__main__":
    main()
