# SPDX-License-Identifier: Apache-2.0
"""Fail-closed local receipt for Foundation maths Term 4 Weeks 35-36."""

from __future__ import annotations

import argparse
import array
import hashlib
import json
import re
import subprocess
import wave
import xml.etree.ElementTree as ET
from collections import Counter
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
ROWS = {
    "AC9MFN06": 17176,
    "AC9MFA01": 17181,
    "AC9MFSP01": 17199,
    "AC9MFSP02": 17205,
}
STEMS = (
    "share-and-group-mat",
    "room-location-mat",
    "repeat-unit-mat",
    "shape-reason-cards",
    "blank-share-cards",
    "practice-cutouts",
    "teacher-held-checks",
    "fresh-station-drawing",
)
EXPECTED_CODES = {
    171: {"AC9MFN06"},
    172: {"AC9MFN06"},
    173: {"AC9MFN06"},
    174: {"AC9MFN06", "AC9MFSP02"},
    175: {"AC9MFN06", "AC9MFSP02"},
    176: {"AC9MFA01"},
    177: {"AC9MFA01"},
    178: {"AC9MFA01"},
    179: {"AC9MFA01", "AC9MFSP01"},
    180: {"AC9MFA01", "AC9MFSP01"},
}
WEEK_SPLIT_DAY = 175
LESSON_MINUTES = 25
FRESH_CHECK_COUNT = 2
MIN_PDF_WORDS = 20
AUDIO_DURATION_MIN = 3.3
AUDIO_DURATION_MAX = 3.5
AUDIO_ACTIVE_THRESHOLD = 250
AUDIO_CLUSTER_GAP = 0.06
AUDIO_CUE_COUNT = 6
AUDIO_UNIT_REPEATS = 3


def need(ok: object, message: str) -> None:
    """Raise a clear failure on missing or drifted authored evidence."""
    if not ok:
        raise AssertionError(message)


def read(name: str) -> str:
    """Read one UTF-8 pack file."""
    return (ROOT / name).read_text(encoding="utf-8")


def source() -> None:
    """Validate the dated official workbook and four exact content rows."""
    workbook = (
        STUDIO
        / "research/sources"
        / "acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    data_path = STUDIO / "data/frameworks/acara-v9.json"
    data = json.loads(data_path.read_text(encoding="utf-8"))
    snapshot = json.loads(read("source-snapshot.json"))
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest() == SOURCE_SHA,
        "Official workbook SHA-256 changed",
    )
    need(
        data["source_sha256"] == snapshot["workbook_sha256"] == SOURCE_SHA,
        "Workbook/import/source hash drift",
    )
    need(snapshot["content_rows"] == ROWS, "Pinned source row map drift")
    records = {
        record["code"]: record
        for record in data["records"]
        if record.get("record_type") == "content_description"
        and record.get("code") in ROWS
    }
    need(set(records) == set(ROWS), "Missing exact ACARA content row")
    crosswalk = read("CURRICULUM-CROSSWALK.md")
    for code, row_number in ROWS.items():
        record = records[code]
        attributes = record["attributes"]
        need(
            record["source_row"] == row_number
            and attributes["level"] == "Foundation Year"
            and attributes["learning_area"] == "Mathematics",
            f"Wrong row, level or learning area: {code}",
        )
        pattern = (
            rf"^\| {code} \| {row_number} \| Mathematics · Foundation Year \| "
            rf"{re.escape(record['plain_text'])} \|"
        )
        need(
            re.search(pattern, crosswalk, re.MULTILINE) is not None,
            f"Exact source wording absent from crosswalk: {code}",
        )
    need(
        "partial" in crosswalk.lower() and "own body position" in crosswalk,
        "Partial coverage and spatial boundary absent",
    )
    print("PASS four exact ACARA v9 Foundation Mathematics rows and partial boundaries")


def pedagogy() -> None:
    """Check the teachable 10-day scripts, choices and optional swaps."""
    lessons = read("LESSONS.md")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", lessons, re.MULTILINE))
    need(
        [int(mark.group(2)) for mark in marks] == list(range(171, 181)),
        "Need Days 171-180 once and in order",
    )
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        end = marks[index + 1].start() if index + 1 < len(marks) else len(lessons)
        body = lessons[mark.end() : end]
        week = 35 if day <= WEEK_SPLIT_DAY else 36
        need(int(mark.group(1)) == week, f"Wrong week at Day {day}")
        times = [
            int(value)
            for value in re.findall(
                r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*",
                body,
                re.MULTILINE,
            )
        ]
        need(
            times == [3, 4, 5, 7, 4, 2] and sum(times) == LESSON_MINUTES,
            f"Day {day} timed routine drift: {times}",
        )
        codes = set(re.findall(r"\bAC9MF(?:N|A|SP)\d\d\b", body))
        need(codes == EXPECTED_CODES[day], f"Day {day} code drift: {codes}")
        need(
            "Evidence check" in body or "Reason check" in body or "Boundary" in body,
            f"Day {day} missing a reason/validity checkpoint",
        )
    cards = read("LEARNER-CARDS.md")
    card_marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need(
        [int(mark.group(1)) for mark in card_marks] == list(range(171, 181)),
        "Need ten clean learner cards",
    )
    for index, mark in enumerate(card_marks):
        end = (
            card_marks[index + 1].start() if index + 1 < len(card_marks) else len(cards)
        )
        body = cards[mark.end() : end]
        need(
            all(f"**{route} ·" in body for route in "ABC")
            and "**Same target:**" in body,
            f"Day {mark.group(1)} choices or shared target absent",
        )
    swaps = read("PRACTICE-SWAPS.md")
    for day in range(171, 181):
        need(
            re.search(rf"^\| {day} \| \*\*[^|]+ \| \*\*[^|]+ \|", swaps, re.MULTILINE),
            f"Day {day} lacks two worked optional swaps",
        )
    need(
        "after-check" in cards.lower() and "not fixed learning styles" in cards,
        "Choice/access/first-check boundary absent",
    )
    print("PASS ten 25-minute scripts, thirty same-target routes and twenty swaps")


def checks() -> None:
    """Check fresh formative cases, separate key and exact arithmetic."""
    student = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    practice = read("MATERIALS.md") + read("PRACTICE-SWAPS.md")
    need(
        len(re.findall(r"^## Check [AB] · Day (?:175|180)", student, re.MULTILINE))
        == FRESH_CHECK_COUNT,
        "Need two fresh check prompts",
    )
    need(
        "teacher/KEY-AND-NEXT.md" not in student,
        "Learner check directly links to public teacher key",
    )
    need(
        "**5 cards each**" in key and "**5 groups of 2 cards**" in key,
        "Check A worked share/group answer drift",
    )
    need(
        "GROUP AREA 1" in key and "GROUP AREA 2" in key,
        "Check A location worked answer drift",
    )
    need(
        "**RECTANGLE, CIRCLE, CIRCLE**" in key
        and "**RECTANGLE\u2013CIRCLE\u2013CIRCLE**" in key,
        "Check B repeating-unit key drift",
    )
    need(
        "ten blank cards" not in practice.lower()
        and "RECTANGLE, CIRCLE, CIRCLE" not in practice,
        "Fresh check set duplicated in routine practice",
    )
    need(
        "**not secure exams**" in student
        and "**not secure exams**" in key
        and "first child-controlled" in student,
        "Public formative/check-first boundary absent",
    )
    expected_share = [3, 4]
    actual_share = [6 // 2, 8 // 2]
    need(actual_share == expected_share, "Practice sharing arithmetic drift")
    actual_group_counts = [9 // 3, 10 // 2]
    need(actual_group_counts == [3, 5], "Grouping/check arithmetic drift")
    print("PASS two fresh public checks, separate key and share/pattern reasoning")


def assets() -> None:
    """Verify A4, text accessibility and exact card geometry."""
    namespace = {"svg": "http://www.w3.org/2000/svg"}
    alternatives = read("print/TEXT-ALTERNATIVES.md")
    for stem in STEMS:
        svg_path = ROOT / "print" / f"{stem}.svg"
        pdf_path = ROOT / "print" / f"{stem}.pdf"
        top = ET.parse(svg_path).getroot()
        need(
            top.get("width") == "210mm" and top.get("height") == "297mm",
            f"{stem} SVG is not physical A4",
        )
        need(
            top.find("svg:title", namespace) is not None
            and top.find("svg:desc", namespace) is not None,
            f"{stem} SVG title/description absent",
        )
        info = subprocess.check_output(["pdfinfo", str(pdf_path)], text=True)
        body = subprocess.check_output(["pdftotext", str(pdf_path), "-"], text=True)
        need(
            re.search(r"Pages:\s+1\b", info) is not None and "(A4)" in info,
            f"{stem} PDF page count/size wrong",
        )
        need(
            "CC BY 4.0" in body and len(body.split()) > MIN_PDF_WORDS,
            f"{stem} PDF selectable text/rights absent",
        )
        need(
            f"{stem}.svg" in alternatives and f"{stem}.pdf" in alternatives,
            f"{stem} full-text alternative absent",
        )
        if stem in {"blank-share-cards", "practice-cutouts"}:
            cards = [
                rect.get("data-card-kind")
                for rect in top.findall(".//svg:rect", namespace)
                if rect.get("data-card-kind") is not None
            ]
            expected = (
                Counter({"blank": 12})
                if stem == "blank-share-cards"
                else Counter({"circle": 4, "triangle": 4, "square": 2, "rectangle": 2})
            )
            need(Counter(cards) == expected, f"{stem} cutout shape counts drift")
        if stem == "room-location-mat":
            need(
                "BETWEEN" not in body and "BESIDE" not in body,
                "Child room mat prints location answers",
            )
        if stem == "fresh-station-drawing":
            need(
                "RECTANGLE" not in body and "CIRCLE" not in body,
                "Child check drawing prints shape answers",
            )
    need(
        "untagged" in alternatives and "tactile" in alternatives.lower(),
        "Print access limits absent",
    )
    print(f"PASS {len(STEMS)} original A4 SVG/PDF aids and exact cutout counts")


def audio() -> None:
    """Verify optional WAV structure and six separated alternating cues."""
    sound = ROOT / "audio/tap-clap-pattern.wav"
    with wave.open(str(sound), "rb") as file:
        channels = file.getnchannels()
        width = file.getsampwidth()
        rate = file.getframerate()
        frames = file.getnframes()
        samples = array.array("h", file.readframes(frames))
    need((channels, width, rate) == (1, 2, 16_000), "WAV format drift")
    need(AUDIO_DURATION_MIN < frames / rate < AUDIO_DURATION_MAX, "WAV duration drift")
    active = []
    for index in range(0, len(samples), 320):
        chunk = samples[index : index + 320]
        if max(abs(value) for value in chunk) > AUDIO_ACTIVE_THRESHOLD:
            active.append(index / rate)
    clusters: list[list[float]] = []
    for time in active:
        if not clusters or time - clusters[-1][-1] > AUDIO_CLUSTER_GAP:
            clusters.append([time])
        else:
            clusters[-1].append(time)
    need(
        len(clusters) == AUDIO_CUE_COUNT,
        f"Expected six separate synthetic cues: {len(clusters)}",
    )
    durations = [cluster[-1] - cluster[0] for cluster in clusters]
    need(
        all(durations[index] < durations[index + 1] for index in (0, 2, 4)),
        "TAP/CLAP alternating short/long cue structure drift",
    )
    transcript = read("audio/TEXT-ALTERNATIVE.md")
    need(
        "The full repeating unit is **TAP\u2013CLAP**" in transcript
        and transcript.count("TAP —") == AUDIO_UNIT_REPEATS
        and transcript.count("CLAP —") == AUDIO_UNIT_REPEATS,
        "Audio exact text/tactile alternative drift",
    )
    print(
        "PASS original 16 kHz WAV with six alternating cues and exact text alternative"
    )


def slug(value: str) -> str:
    """Approximate GitHub Markdown heading anchors used by local files."""
    value = re.sub(r"\[[^]]+\]\([^)]+\)", "", value)
    value = re.sub(r"[`*_]", "", value).strip().lower()
    value = re.sub(r"[^\w -]", "", value)
    return value.replace(" ", "-")


def links(*, allow_manifest_bootstrap: bool = False) -> int:
    """Fail on missing local files and heading anchors."""
    total = 0
    for path in ROOT.rglob("*.md"):
        content = path.read_text(encoding="utf-8")
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", content):
            if url.startswith(("https://", "http://", "mailto:")):
                continue
            name, marker, anchor = unquote(url).partition("#")
            target = (path.parent / name).resolve() if name else path
            if allow_manifest_bootstrap and target == ROOT / "manifest.json":
                total += 1
                continue
            need(target.is_file(), f"Broken local link in {path}: {url}")
            if marker and target.suffix.lower() == ".md":
                headings = [
                    slug(title)
                    for title in re.findall(
                        r"^#{1,6} (.+)$",
                        target.read_text(encoding="utf-8"),
                        re.MULTILINE,
                    )
                ]
                need(anchor in headings, f"Broken heading anchor in {path}: {url}")
            total += 1
    print(f"PASS {total} local file and heading links")
    return total


def manifest_data() -> dict:
    """Hash every authored file except the manifest itself."""
    files = sorted(
        path
        for path in ROOT.rglob("*")
        if path.is_file()
        and path.name != "manifest.json"
        and "__pycache__" not in path.parts
    )
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/mathematics/term-4/weeks-35-36",
        "created_at": "2026-09-29",
        "review_status": (
            "author_desk_checked_pending_educator_child_"
            "accessibility_local_syllabus_review"
        ),
        "curriculum_source_sha256": SOURCE_SHA,
        "curriculum_codes": sorted(ROWS),
        "files": {
            str(path.relative_to(ROOT)): {
                "sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
                "bytes": path.stat().st_size,
            }
            for path in files
        },
    }


def main() -> None:
    """Run all offline receipts and optionally freeze hashes."""
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    source()
    pedagogy()
    checks()
    assets()
    audio()
    links(allow_manifest_bootstrap=args.write_manifest)
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n",
            encoding="utf-8",
        )
        print(f"WROTE SHA-256 manifest for {len(expected['files'])} files")
    else:
        need(
            json.loads(manifest.read_text(encoding="utf-8")) == expected,
            "SHA-256 manifest missing or stale",
        )
        print(f"PASS SHA-256 manifest for {len(expected['files'])} files")
    print("PACK PASS · authored offline artifacts; no live classroom/access validation")


if __name__ == "__main__":
    main()
