#!/usr/bin/env python3
"""Read-only exact-row, teaching, print, sound and hash audit for T3 W21–22."""

from __future__ import annotations

import argparse
import array
import hashlib
import json
import math
import re
import subprocess
import unicodedata
import wave
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
ACARA = ROOT.parents[4] / "data/frameworks/acara-v9.json"
WORKBOOK_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
QCAA = "https://www.qcaa.qld.edu.au/p-10/aciq/version-9/learning-areas/p-10-the-arts/music"
CODES = {
    "AC9AMUFE01": (21218, "explore how and why the arts are important for people and communities"),
    "AC9AMUFD01": (21227, "use play, imagination, arts knowledge, processes and/or skills to discover possibilities and develop ideas"),
    "AC9AMUFC01": (21237, "create arts works that communicate ideas"),
    "AC9AMUFP01": (21246, "share their arts works with audiences"),
}
AIDS = ("section-road", "exact-return", "contrast-return", "compare-endings", "plan-actual", "transfer-key", "make-share")
CHECKS = ("return-1", "return-2")
STARTS = (.10, .45, 1.05, 1.40, 2.00, 2.35)
CUES = {
    "cue-return-lh": ("L", "H", "H", "H", "L", "H"),
    "cue-return-hl": ("H", "L", "L", "L", "H", "L"),
    "cue-end-different": ("L", "H", "H", "H", "H", "L"),
}
REQUIRED = {
    "README.md", "CURRICULUM-CROSSWALK.md", "SOURCE-AND-RIGHTS.md",
    "AUTHENTIC-SOURCE-SLOT.md", "MATERIALS.md", "LESSONS.md",
    "LEARNER-CARDS.md", "PRACTICE-SWAPS.md", "EXAMPLE-BANK.md",
    "PROMPTS.md", "STUDENT-CHECKS.md", "teacher/KEY-AND-NEXT.md",
    "RUN-THROUGH.md", "audio/ASSET-INDEX.md", "print/TEXT-ALTERNATIVES.md",
    "print/FONT-RIGHTS.md", "print/dejavu-font-copyright.txt",
    "generate_assets.py", "verify_pack.py",
}
for folder, stems in (("print", AIDS), ("check", CHECKS)):
    for stem in stems:
        REQUIRED.update((f"{folder}/{stem}.svg", f"{folder}/{stem}.pdf"))
for name in CUES:
    REQUIRED.add(f"audio/{name}.wav")


def need(condition: bool, message: str) -> None:
    if not condition:
        raise AssertionError(message)


def get(relative: str) -> str:
    return (ROOT / relative).read_text(encoding="utf-8")


def row(relative: str, day: int, count: int) -> list[str]:
    matches = re.findall(rf"^\| {day} \|(.+)$", get(relative), re.MULTILINE)
    need(len(matches) == 1, f"{relative}: day {day} missing/duplicated")
    cells = [part.strip() for part in matches[0].strip().strip("|").split("|")]
    need(len(cells) == count and all(cells), f"{relative}: day {day} empty/wrong cells")
    return cells


def curriculum() -> None:
    data = json.loads(ACARA.read_text(encoding="utf-8"))
    need(data["source_sha256"] == WORKBOOK_SHA, "ACARA workbook source SHA changed")
    records = {r.get("code"): r for r in data["records"] if r.get("code") in CODES}
    cross = get("CURRICULUM-CROSSWALK.md")
    need(QCAA in cross and WORKBOOK_SHA in cross and "Prep" in cross and "403" in cross, "official pin/retrieval caveat missing")
    for code, (source_row, wording) in CODES.items():
        record = records.get(code)
        need(record is not None and record["source_row"] == source_row and record["plain_text"] == wording, f"{code}: source row drift")
        need(code in cross and str(source_row) in cross and wording in cross, f"{code}: crosswalk drift")
    for word in ("partial", "held", "audience", "actual"):
        need(word in cross.lower(), f"crosswalk boundary absent: {word}")
    rights = get("SOURCE-AND-RIGHTS.md").lower()
    for word in ("cc by 4.0", "candidate", "dejavu", "first nations", "no child data"):
        need(word in rights, f"rights absent: {word}")
    slot = get("AUTHENTIC-SOURCE-SLOT.md").lower()
    need("permission" in slot and "first nations" in slot and "maker" in slot, "real-work gate absent")
    readme = get("README.md").lower()
    for word in ("term 3", "no-sound", "human", "partial", "audience"):
        need(word in readme, f"README scope absent: {word}")


def teaching() -> None:
    lessons = get("LESSONS.md")
    for day in range(101, 111):
        week = 21 if day <= 105 else 22
        blocks = re.findall(rf"^## Week {week} · Day {day} ·.*?(?=^## Week |\Z)", lessons, re.MULTILINE | re.DOTALL)
        need(len(blocks) == 1, f"day {day}: lesson absent/duplicated")
        block = blocks[0]
        need("**Target/codes:**" in block and "**Prepare:**" in block, f"day {day}: target/prep")
        stages = (
            ("Welcome", "Access", "Independent plan", "Child response", "Self-check", "Close")
            if day in (105, 110)
            else ("Welcome", "Model", "Try together", "Child choice", "Notice", "Close")
        )
        durations = []
        for stage in stages:
            found = re.findall(rf"\*\*{re.escape(stage)} · (\d+) min\.\*\*", block)
            need(len(found) == 1, f"day {day}: missing {stage}")
            durations.append(int(found[0]))
        need(durations == [3, 4, 5, 7, 4, 2], f"day {day}: not 25 min")
        routes = row("LEARNER-CARDS.md", day, 3)
        need(len(set(routes)) == 3, f"day {day}: repeated route")
        swaps = row("PRACTICE-SWAPS.md", day, 2)
        need(all("→" in x for x in swaps), f"day {day}: unworked swap")
        row("EXAMPLE-BANK.md", day, 2)
        row("PROMPTS.md", day, 1)
        row("teacher/KEY-AND-NEXT.md", day, 2)
    for day in (105, 110):
        need(all("After Check" in x for x in row("PRACTICE-SWAPS.md", day, 2)), f"day {day}: after-check hold absent")
    materials = get("MATERIALS.md")
    for letter in "ABCDEFG":
        need(len(re.findall(rf"^### {letter} ·", materials, re.MULTILINE)) == 1, f"practice {letter} absent")
    before_one = lessons.split("## Week 21 · Day 105", maxsplit=1)[0]
    need("H H | L H | H H" not in before_one, "Return 1 score leaked before check")
    before_two = lessons.split("## Week 22 · Day 110", maxsplit=1)[0]
    need("A B | B B" not in before_two, "Return 2 source leaked before check")
    public = get("STUDENT-CHECKS.md")
    key = get("teacher/KEY-AND-NEXT.md")
    need("Check Return 1" in public and "Check Return 2" in public, "public checks missing")
    for stem in CHECKS:
        need(f"check/{stem}.svg" in public and f"check/{stem}.pdf" in public, f"{stem}: printable links")
    need("third section RETURN = H H" in key and "final pair must be **A B**" in key, "teacher key arithmetic")
    need("teacher/KEY-AND-NEXT.md" not in public and "KEY-AND-NEXT.md" not in get("LEARNER-CARDS.md"), "private key linked publicly")
    need("third section RETURN" not in get("check/return-1.svg"), "Return 1 answer leaked")
    need("final pair must" not in get("check/return-2.svg"), "Return 2 answer leaked")


def print_media() -> None:
    alternatives = get("print/TEXT-ALTERNATIVES.md")
    for folder, stems in (("print", AIDS), ("check", CHECKS)):
        for stem in stems:
            root = ET.parse(ROOT / folder / f"{stem}.svg").getroot()
            need(root.attrib.get("width") == "210mm" and root.attrib.get("height") == "297mm" and root.attrib.get("viewBox") == "0 0 794 1123", f"{stem}: SVG A4")
            need(root.attrib.get("role") == "img" and root.attrib.get("aria-labelledby") == "title desc", f"{stem}: SVG semantics")
            tags = [x.tag.rsplit("}", maxsplit=1)[-1] for x in root.iter()]
            need("title" in tags and "desc" in tags and tags.count("rect") >= 5 and tags.count("text") >= 12, f"{stem}: graphics/text absent")
            printed = [x.text or "" for x in root.iter() if x.tag.rsplit("}", maxsplit=1)[-1] == "text"]
            need(" | ".join(printed) in alternatives and f"{folder}/{stem}.svg" in alternatives, f"{stem}: exact alternative drift")
            info = subprocess.check_output(["pdfinfo", str(ROOT / folder / f"{stem}.pdf")], text=True)
            need(re.search(r"Pages:\s+1\b", info) is not None and re.search(r"Page size:\s+595\.\d+ x 841\.\d+ pts \(A4\)", info) is not None, f"{stem}: PDF A4")
            searchable = subprocess.check_output(["pdftotext", "-layout", str(ROOT / folder / f"{stem}.pdf"), "-"], text=True)
            need(len(searchable.strip()) > 140, f"{stem}: PDF text absent")
            fonts = subprocess.check_output(["pdffonts", str(ROOT / folder / f"{stem}.pdf")], text=True)
            need("DejaVu" in fonts and "yes" in fonts, f"{stem}: font not embedded")
    need("bitstream" in get("print/dejavu-font-copyright.txt").lower(), "font licence absent")


def rms(samples: list[int]) -> float:
    return math.sqrt(sum(float(v) ** 2 for v in samples) / len(samples))


def strength(samples: list[int], hz: int) -> float:
    n = len(samples)
    c = sum(v * math.cos(2 * math.pi * hz * i / 22050) for i, v in enumerate(samples))
    s = sum(v * math.sin(2 * math.pi * hz * i / 22050) for i, v in enumerate(samples))
    return 2 * math.hypot(c, s) / n


def audio() -> None:
    index = get("audio/ASSET-INDEX.md").lower()
    for word in ("candidate", "not cleared", "human educator", "volume", "no-sound", "0.10", "2.35", "330 hz", "392 hz"):
        need(word in index, f"audio transcript/hold absent: {word}")
    for name, values in CUES.items():
        need(f"{name}.wav" in index, f"{name}: transcript absent")
        with wave.open(str(ROOT / "audio" / f"{name}.wav"), "rb") as handle:
            need(handle.getnchannels() == 1 and handle.getsampwidth() == 2 and handle.getframerate() == 22050 and handle.getnframes() == round(2.9 * 22050), f"{name}: PCM format")
            data = array.array("h")
            data.frombytes(handle.readframes(handle.getnframes()))
        need(max(abs(x) for x in data) < 1000, f"{name}: digital peak")
        need(rms(data[:round(.08 * 22050)]) < 2 and rms(data[round(2.6 * 22050):]) < 2, f"{name}: silence ends")
        for event, (start, key) in enumerate(zip(STARTS, values, strict=True), 1):
            samples = data[round((start + .03) * 22050):round((start + .13) * 22050)]
            need(520 < rms(samples) < 640, f"{name}: event {event} level")
            expected, other = (330, 392) if key == "L" else (392, 330)
            need(strength(samples, expected) > 700 and strength(samples, other) < 160, f"{name}: event {event} pitch")
            gap = data[round((start + .20) * 22050):round((start + .27) * 22050)]
            need(rms(gap) < 2, f"{name}: event {event} tail")


def slug(heading: str) -> str:
    lower = heading.lower()
    return "".join(c for c in lower if c in "-_ " or unicodedata.category(c)[0] in "LN").replace(" ", "-")


def links() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        body = path.read_text(encoding="utf-8")
        for match in re.finditer(r"(?<!!)\[[^\]]+\]\(([^)]+)\)", body):
            ref = unquote(match.group(1))
            if ref.startswith(("https://", "http://", "mailto:")):
                continue
            name, _, anchor = ref.partition("#")
            target = path if not name else (path.parent / name).resolve()
            need(target.is_file(), f"broken local link {path.relative_to(ROOT)} → {ref}")
            if anchor and target.suffix == ".md":
                headings = {slug(h) for h in re.findall(r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)}
                need(anchor in headings, f"broken heading {path.relative_to(ROOT)} → {ref}")
            count += 1
        lines = body.splitlines()
        for i, item in enumerate(lines):
            if re.match(r"^\s*(?:[-*+] |\d+\. )", item) and i > 0:
                previous = lines[i - 1]
                if not re.match(r"^\s*(?:[-*+] |\d+\. )", previous):
                    need(not previous.strip(), f"{path.relative_to(ROOT)}:{i+1}: list spacing")
    return count


def hashes() -> dict[str, str]:
    return {
        str(p.relative_to(ROOT)): hashlib.sha256(p.read_bytes()).hexdigest()
        for p in sorted(ROOT.rglob("*"))
        if p.is_file() and p.name != "ASSET-MANIFEST.json" and "__pycache__" not in p.parts
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    present = set(hashes())
    need(REQUIRED <= present, f"missing files: {sorted(REQUIRED - present)}")
    curriculum()
    teaching()
    print_media()
    audio()
    link_count = links()
    manifest = {
        "schema": "subjectnest-pack-sha256-v1",
        "pack": "foundation/music/term-3/weeks-21-22",
        "scope": "ten optional 25-minute Prep Music starts; partial ACARA v9 alignment",
        "files": hashes(),
    }
    target = ROOT / "ASSET-MANIFEST.json"
    if args.write_manifest:
        target.write_text(json.dumps(manifest, indent=2, sort_keys=True) + "\n", encoding="utf-8")
    else:
        need(target.is_file(), "manifest missing; write intentionally after QA")
        need(json.loads(target.read_text(encoding="utf-8")) == manifest, "manifest SHA drift")
    print(
        "PASS: 10 x 25-minute scripts; 30 routes; 20 swaps; 20 bridges; "
        "2 fresh checks/keys; 9 A4 pairs; 3 candidate cues; "
        f"{link_count} local links; {len(manifest['files'])} SHA-256 files"
    )


if __name__ == "__main__":
    main()
