# SPDX-License-Identifier: Apache-2.0
"""Fail-closed offline receipt for Foundation maths Term 4 Weeks 39-40."""

from __future__ import annotations

import argparse
import ast
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from collections import Counter
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
QCAA_URL = (
    "https://www.qcaa.qld.edu.au/p-10/aciq/version-9/learning-areas/p-10-mathematics"
)
ROWS = {
    "AC9MFN01": 17151,
    "AC9MFN02": 17157,
    "AC9MFN03": 17160,
    "AC9MFN04": 17166,
    "AC9MFN05": 17171,
    "AC9MFN06": 17176,
    "AC9MFA01": 17181,
    "AC9MFM01": 17187,
    "AC9MFM02": 17192,
    "AC9MFSP01": 17199,
    "AC9MFSP02": 17205,
    "AC9MFST01": 17211,
}
STEMS = (
    "zero-to-twenty-rail",
    "part-whole-mat",
    "shape-and-place-mat",
    "given-data-and-evidence-mat",
    "show-check-explain-mat",
    "repeat-and-reason-strip",
    "direct-length-and-day-parts",
    "check-a-explanation-source",
    "check-b-transfer-source",
)
EXPECTED_CODES = {
    191: {"AC9MFN01", "AC9MFN03"},
    192: {"AC9MFN02", "AC9MFN04"},
    193: {"AC9MFSP01", "AC9MFSP02"},
    194: {"AC9MFST01", "AC9MFN03"},
    195: {"AC9MFN01", "AC9MFN03", "AC9MFSP01", "AC9MFSP02", "AC9MFST01"},
    196: {"AC9MFN01", "AC9MFN04", "AC9MFN05"},
    197: {"AC9MFA01", "AC9MFSP01", "AC9MFSP02"},
    198: {"AC9MFM01", "AC9MFM02"},
    199: {"AC9MFN06", "AC9MFST01", "AC9MFN03"},
    200: {"AC9MFN04", "AC9MFN05", "AC9MFA01", "AC9MFM01"},
}
WEEK_SPLIT_DAY = 195
LESSON_MINUTES = 25
MIN_PDF_WORDS = 18


def need(ok: object, message: str) -> None:
    """Fail with an actionable artifact description."""
    if not ok:
        raise AssertionError(message)


def read(name: str) -> str:
    """Read authored UTF-8 material."""
    return (ROOT / name).read_text(encoding="utf-8")


def source() -> None:
    """Pin workbook bytes, imported rows, exact wording and Queensland entry."""
    workbook = (
        STUDIO
        / "research/sources"
        / "acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    data = json.loads(
        (STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8")
    )
    snapshot = json.loads(read("source-snapshot.json"))
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest() == SOURCE_SHA,
        "Official workbook hash drift",
    )
    need(
        data["source_sha256"] == snapshot["workbook_sha256"] == SOURCE_SHA,
        "Import/source hash drift",
    )
    need(snapshot["content_rows"] == ROWS, "Pinned source rows drift")
    need(
        snapshot["queensland_prep_entry_point_url"] == QCAA_URL,
        "QCAA entry point drift",
    )
    crosswalk = read("CURRICULUM-CROSSWALK.md")
    need(
        QCAA_URL in crosswalk and "partial" in crosswalk.lower(),
        "Queensland/partial boundary absent",
    )
    records = {
        record["code"]: record
        for record in data["records"]
        if record.get("record_type") == "content_description"
        and record.get("code") in ROWS
    }
    need(set(records) == set(ROWS), "Missing exact ACARA Foundation row")
    for code, row_number in ROWS.items():
        record = records[code]
        attrs = record["attributes"]
        need(
            record["source_row"] == row_number
            and attrs["level"] == "Foundation Year"
            and attrs["learning_area"] == "Mathematics",
            f"Wrong level/area/row: {code}",
        )
        pattern = (
            rf"^\| {code} \| {row_number} \| Mathematics · Foundation Year \| "
            rf"{re.escape(record['plain_text'])} \|"
        )
        need(
            re.search(pattern, crosswalk, re.MULTILINE),
            f"Verbatim wording absent: {code}",
        )
    print("PASS 12 exact ACARA v9 Foundation Mathematics rows and Queensland boundary")


def pedagogy() -> None:
    """Check ten scripts, 30 same-target routes and 20 worked transfers."""
    lessons = read("LESSONS.md")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", lessons, re.MULTILINE))
    need(
        [int(mark.group(2)) for mark in marks] == list(range(191, 201)),
        "Day sequence 191-200 drift",
    )
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        end = marks[index + 1].start() if index + 1 < len(marks) else len(lessons)
        body = lessons[mark.end() : end]
        need(
            int(mark.group(1)) == (39 if day <= WEEK_SPLIT_DAY else 40),
            f"Wrong week Day {day}",
        )
        times = [
            int(value)
            for value in re.findall(
                r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*",
                body,
                re.MULTILINE,
            )
        ]
        need(
            times == [3, 4, 5, 7, 4, 2] and sum(times) == LESSON_MINUTES,
            f"Day {day} timed routine drift: {times}",
        )
        codes = set(re.findall(r"\bAC9MF(?:N|A|M|SP|ST)\d\d\b", body))
        need(codes == EXPECTED_CODES[day], f"Day {day} code drift: {codes}")
        need("check" in body.lower(), f"Day {day} lacks an evidence check")
    cards = read("LEARNER-CARDS.md")
    card_marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need(
        [int(mark.group(1)) for mark in card_marks] == list(range(191, 201)),
        "Learner card sequence drift",
    )
    for index, mark in enumerate(card_marks):
        end = (
            card_marks[index + 1].start() if index + 1 < len(card_marks) else len(cards)
        )
        body = cards[mark.end() : end]
        need(
            all(re.search(rf"^\- \*\*{route} ·", body, re.MULTILINE) for route in "ABC")
            and "**Same target:**" in body
            and "**Home:**" in body,
            f"Three routes/shared target/home missing Day {mark.group(1)}",
        )
    swaps = read("PRACTICE-SWAPS.md")
    for day in range(191, 201):
        need(
            re.search(rf"^\| {day} \| \*\*[^|]+ \| \*\*[^|]+ \|", swaps, re.MULTILINE),
            f"Two worked swaps absent Day {day}",
        )
    need(
        "not fixed learning styles" in cards and "After-check practice only" in cards,
        "Access/hold boundary missing",
    )
    print("PASS ten 25-minute scripts, 30 switchable routes and 20 worked swaps")


def checks() -> None:
    """Check new public cases, separate key and internal arithmetic."""
    student = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    practice = read("PRACTICE-SWAPS.md") + read("MATERIALS.md")
    need(
        re.findall(r"^## Check ([AB]) · Day (\d+)\b", student, re.MULTILINE)
        == [("A", "195"), ("B", "200")],
        "Fresh check A/B order drift",
    )
    need("teacher/KEY-AND-NEXT.md" not in student, "Learner check links to teacher key")
    need(
        "not secure exams" in student and "public by URL" in key,
        "Public formative boundary absent",
    )
    need(
        "first child-controlled" in student and "not available" in student.lower(),
        "First response/handover boundary absent",
    )
    for phrase in (
        "**16 separate broad round marks**",
        "**one too high**",
        "**above fixed BOARD**",
        "**9 PINE**",
        "**5 SHELL**",
        "**2 + 6 = 8**",
        "**7 + 3 = 10**",
        "**SQUARE\u2013CIRCLE\u2013CIRCLE**",
        "**B is longer**",
    ):
        need(phrase in key, f"Worked key phrase/value drift: {phrase}")
    need(
        "9 PINE" not in practice
        and "5 SHELL" not in practice
        and "7 + 3" not in practice
        and "SQUARE\u2013CIRCLE\u2013CIRCLE" not in practice,
        "Held source duplicated in routine practice",
    )
    actual = [9 + 5, 9 - 5, 2 + 6, 7 + 3, 6 + 4, 6 + 2, 8 - 3, 10 // 2]
    expected = [14, 4, 8, 10, 10, 8, 5, 5]
    need(actual == expected, "Independent check/practice arithmetic drift")
    print("PASS two fresh public checks, separate key and arithmetic")


def count_attrs(top: ET.Element, tag: str, attribute: str) -> Counter:
    """Count machine-readable SVG source attributes."""
    namespace = {"svg": "http://www.w3.org/2000/svg"}
    return Counter(
        item.get(attribute)
        for item in top.findall(f".//svg:{tag}", namespace)
        if item.get(attribute) is not None
    )


def assets() -> None:
    """Check physical A4, selectable PDFs and exact fixed source geometry."""
    namespace = {"svg": "http://www.w3.org/2000/svg"}
    alternatives = read("print/TEXT-ALTERNATIVES.md")
    tree = ast.parse(read("print/generate_print.py"))
    page_map = next(
        node.value
        for node in tree.body
        if isinstance(node, ast.Assign)
        and any(
            isinstance(target, ast.Name) and target.id == "PAGES"
            for target in node.targets
        )
    )
    need(
        isinstance(page_map, ast.Dict)
        and {key.value for key in page_map.keys if isinstance(key, ast.Constant)}
        == set(STEMS),
        "Generator/stem inventory mismatch",
    )
    for stem in STEMS:
        svg_path = ROOT / "print" / f"{stem}.svg"
        pdf_path = ROOT / "print" / f"{stem}.pdf"
        top = ET.parse(svg_path).getroot()
        need(
            top.get("width") == "210mm" and top.get("height") == "297mm",
            f"SVG is not A4: {stem}",
        )
        need(
            top.find("svg:title", namespace) is not None
            and top.find("svg:desc", namespace) is not None,
            f"SVG metadata absent: {stem}",
        )
        info = subprocess.check_output(["pdfinfo", str(pdf_path)], text=True)
        body = subprocess.check_output(["pdftotext", str(pdf_path), "-"], text=True)
        need(
            re.search(r"Pages:\s+1\b", info) and "(A4)" in info,
            f"PDF page/size drift: {stem}",
        )
        need(
            "CC BY 4.0" in body and len(body.split()) >= MIN_PDF_WORDS,
            f"PDF selectable text absent: {stem}",
        )
        need(
            f"{stem}.svg" in alternatives and f"{stem}.pdf" in alternatives,
            f"Full text alternative absent: {stem}",
        )
        if stem == "zero-to-twenty-rail":
            need(
                count_attrs(top, "rect", "data-number")
                == Counter({str(number): 1 for number in range(21)}),
                "0-20 numeral cells drift",
            )
        elif stem == "repeat-and-reason-strip":
            need(
                count_attrs(top, "rect", "data-pattern-kind")
                == Counter({"CIRCLE": 2, "SQUARE": 2, "TRIANGLE": 2, "?": 3}),
                "Practice repeating-unit/blank slots drift",
            )
        elif stem == "direct-length-and-day-parts":
            need(
                count_attrs(top, "rect", "data-day-part")
                == Counter(
                    {name: 1 for name in ("MORNING", "LUNCHTIME", "AFTERNOON", "NIGHT")}
                ),
                "Day-part cards drift",
            )
            need(
                count_attrs(top, "g", "data-length-units")
                == Counter({"180": 1, "250": 1}),
                "Practice strip length drift",
            )
        elif stem == "check-a-explanation-source":
            need(
                count_attrs(top, "circle", "data-source") == Counter({"NUMBER": 16}),
                "Check A count marks drift",
            )
            need(
                count_attrs(top, "circle", "data-shape") == Counter({"circle": 1}),
                "Check A shape drift",
            )
            need(
                count_attrs(top, "rect", "data-landmark") == Counter({"BOARD": 1}),
                "Check A landmark drift",
            )
            need(
                count_attrs(top, "rect", "data-card-kind")
                == Counter({"PINE": 9, "SHELL": 5}),
                "Check A picture counts drift",
            )
            need(
                count_attrs(top, "g", "data-icon-kind")
                == Counter({"PINE": 9, "SHELL": 5}),
                "Check A actual pictograms drift",
            )
            need(
                "claim 17" in body and "9 PINE" not in body,
                "Check A page prints an answer",
            )
        elif stem == "check-b-transfer-source":
            need(
                count_attrs(top, "rect", "data-part") == Counter({"A": 2, "B": 6}),
                "Check B parts drift",
            )
            need(
                count_attrs(top, "circle", "data-add-group")
                == Counter({"START": 7, "ADD": 3}),
                "Check B add model drift",
            )
            need(
                count_attrs(top, "rect", "data-pattern-kind")
                == Counter({"SQUARE": 2, "CIRCLE": 4, "?": 3}),
                "Check B unit drift",
            )
            need(
                count_attrs(top, "g", "data-length-units")
                == Counter({"145": 1, "230": 1}),
                "Check B strip geometry drift",
            )
            need(
                "claim 9" in body and "7 + 3 = 10" not in body,
                "Check B page prints an answer",
            )
    need(
        "untagged" in alternatives and "tactile" in alternatives.lower(),
        "Print access boundary absent",
    )
    print("PASS nine original A4 SVG/PDF aids and exact source geometry")


def slug(value: str) -> str:
    """Approximate local Markdown heading anchors."""
    value = re.sub(r"\[[^]]+\]\([^)]+\)", "", value)
    value = re.sub(r"[`*_]", "", value).strip().lower()
    value = re.sub(r"[^\w -]", "", value)
    return value.replace(" ", "-")


def links(*, allow_manifest_bootstrap: bool = False) -> int:
    """Check local paths and heading targets without following remote URLs."""
    count = 0
    for path in ROOT.rglob("*.md"):
        body = path.read_text(encoding="utf-8")
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", body):
            if url.startswith(("https://", "http://", "mailto:")):
                continue
            name, marker, anchor = unquote(url).partition("#")
            target = (path.parent / name).resolve() if name else path
            if allow_manifest_bootstrap and target == ROOT / "manifest.json":
                count += 1
                continue
            need(
                target.is_file(), f"Broken local link {path.relative_to(ROOT)} -> {url}"
            )
            if marker and target.suffix.lower() == ".md":
                headings = [
                    slug(title)
                    for title in re.findall(
                        r"^#{1,6} (.+)$",
                        target.read_text(encoding="utf-8"),
                        re.MULTILINE,
                    )
                ]
                need(anchor in headings, f"Broken heading anchor: {url}")
            count += 1
    print(f"PASS {count} local file and heading links")
    return count


def manifest_data() -> dict:
    """Hash every authored file except the manifest itself."""
    files = sorted(
        path
        for path in ROOT.rglob("*")
        if path.is_file()
        and path.name != "manifest.json"
        and "__pycache__" not in path.parts
    )
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/mathematics/term-4/weeks-39-40",
        "created_at": "2026-09-29",
        "review_status": (
            "author_desk_checked_pending_educator_child_"
            "accessibility_local_syllabus_review"
        ),
        "curriculum_source_sha256": SOURCE_SHA,
        "curriculum_codes": sorted(ROWS),
        "files": {
            str(path.relative_to(ROOT)): {
                "sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
                "bytes": path.stat().st_size,
            }
            for path in files
        },
    }


def main() -> None:
    """Run all offline gates and optionally freeze file hashes."""
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    source()
    pedagogy()
    checks()
    assets()
    links(allow_manifest_bootstrap=args.write_manifest)
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n",
            encoding="utf-8",
        )
        print(f"WROTE SHA-256 manifest for {len(expected['files'])} files")
    else:
        need(
            json.loads(manifest.read_text(encoding="utf-8")) == expected,
            "SHA-256 manifest missing or stale",
        )
        print(f"PASS SHA-256 manifest for {len(expected['files'])} files")
    print("PACK PASS · authored offline artifacts; no live classroom/access validation")


if __name__ == "__main__":
    main()
