#!/usr/bin/env python3
"""Read-only-by-default audit of Foundation Drama Weeks 7–8."""

from __future__ import annotations

import argparse
import ast
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
ROWS = {
    "AC9ADRFE01": 20732,
    "AC9ADRFD01": 20741,
    "AC9ADRFC01": 20751,
    "AC9ADRFP01": 20760,
}
AIDS = (
    "now-later",
    "same-changed",
    "scene-shift",
    "boundary-choice",
    "returning-mark",
    "returning-action",
    "two-scene-record",
)
CHECKS = ("fresh-m", "fresh-n")
REQUIRED = {
    "README.md",
    "source-snapshot.json",
    "CODE-LICENSE.txt",
    "CURRICULUM-CROSSWALK.md",
    "AUTHENTIC-SOURCE-SLOT.md",
    "SOURCE-AND-RIGHTS.md",
    "MATERIALS.md",
    "LESSONS.md",
    "LEARNER-CARDS.md",
    "PRACTICE-SWAPS.md",
    "EXAMPLE-BANK.md",
    "PROMPTS.md",
    "STUDENT-CHECKS.md",
    "teacher/KEY-AND-NEXT.md",
    "RUN-THROUGH.md",
    "generate_assets.py",
    "verify_pack.py",
    "print/TEXT-ALTERNATIVES.md",
    "print/dejavu-font-copyright.txt",
}
for stem in AIDS:
    REQUIRED.update((f"print/{stem}.svg", f"print/{stem}.pdf"))
for stem in CHECKS:
    REQUIRED.update((f"check/{stem}.svg", f"check/{stem}.pdf"))


def need(ok: object, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def read(name: str) -> str:
    return (ROOT / name).read_text(encoding="utf-8")


def source() -> None:
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    imported = json.loads(
        (STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8")
    )
    snapshot = json.loads(read("source-snapshot.json"))
    crosswalk = read("CURRICULUM-CROSSWALK.md")
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == imported["source_sha256"]
        == snapshot["workbook_sha256"]
        == SOURCE_SHA,
        "Official workbook, import or source SHA drift",
    )
    need(snapshot["content_rows"] == ROWS, "Official row pin drift")
    for field in (
        "official_workbook_url",
        "queensland_prep_alignment_url",
        "queensland_prep_techniques_url",
    ):
        need(snapshot[field] in crosswalk, f"Official source URL missing: {field}")
    records = {
        record["code"]: record
        for record in imported["records"]
        if record.get("record_type") == "content_description"
        and record.get("code") in ROWS
    }
    need(set(records) == set(ROWS), "Official content-description set drift")
    for code, number in ROWS.items():
        record = records[code]
        attrs = record["attributes"]
        need(
            record["source_row"] == number
            and attrs["level"] == "Foundation Year"
            and attrs["learning_area"] == "The Arts"
            and attrs["subject"] == "Drama",
            f"Official placement drift: {code}",
        )
        expected = rf"^\| {code} \| {number} \| {re.escape(record['plain_text'])} \|"
        need(
            re.search(expected, crosswalk, re.MULTILINE),
            f"Exact official row absent: {code}",
        )
    for phrase in (
        "NOT OBSERVED",
        "not a QCAA instrument",
        "Paper devising",
        "unverified",
    ):
        need(
            phrase.lower() in crosswalk.lower(),
            f"Source/evidence boundary missing: {phrase}",
        )
    print(
        "PASS pinned ACARA workbook SHA, exact four Foundation rows and QCAA Prep URLs"
    )


def table_row(name: str, day: int, cells: int) -> list[str]:
    matches = re.findall(rf"^\| {day} \|(.+)$", read(name), re.MULTILINE)
    need(len(matches) == 1, f"{name}: Day {day} absent or duplicate")
    parts = [part.strip() for part in matches[0].strip().strip("|").split("|")]
    need(len(parts) == cells and all(parts), f"{name}: Day {day} row drift")
    return parts


def daily() -> None:
    lessons = read("LESSONS.md")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", lessons, re.MULTILINE))
    need(
        [int(mark.group(2)) for mark in marks] == list(range(31, 41)),
        "Ten-day inventory drift",
    )
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        week = int(mark.group(1))
        end = marks[index + 1].start() if index + 1 < len(marks) else len(lessons)
        body = lessons[mark.end() : end]
        stages = (
            (
                "Welcome",
                "Access",
                "Independent plan",
                "Child response",
                "Self-check",
                "Close",
            )
            if day in (35, 40)
            else ("Welcome", "Model", "Try together", "Child choice", "Notice", "Close")
        )
        durations = []
        for stage in stages:
            matches = re.findall(rf"\*\*{re.escape(stage)} · (\d+) min\.\*\*", body)
            need(len(matches) == 1, f"Day {day} stage missing/duplicate: {stage}")
            durations.append(int(matches[0]))
        need(
            week == (7 if day <= 35 else 8) and durations == [3, 4, 5, 7, 4, 2],
            f"Day {day} week/timing drift",
        )
        need(
            "**Target/codes:**" in body and "**Prepare:**" in body,
            f"Day {day} target/material drift",
        )
        need(
            "AC9ADRFD01" in body and "AC9ADRFC01" in body,
            f"Day {day} primary code drift",
        )
        need(
            "Save" in body or "Record" in body or "Log" in body,
            f"Day {day} first-evidence move missing",
        )
        routes = table_row("LEARNER-CARDS.md", day, 3)
        need(
            all(len(route) >= 70 for route in routes),
            f"Day {day} alternate route too thin",
        )
        swaps = table_row("PRACTICE-SWAPS.md", day, 2)
        need(
            all("→" in swap and len(swap) >= 70 for swap in swaps),
            f"Day {day} two worked swaps absent",
        )
        bridges = table_row("EXAMPLE-BANK.md", day, 2)
        need(
            all(len(bridge) >= 45 for bridge in bridges),
            f"Day {day} two interest bridges absent",
        )
        table_row("PROMPTS.md", day, 1)
        table_row("teacher/KEY-AND-NEXT.md", day, 2)
    need(
        "not fixed learning styles" in read("LEARNER-CARDS.md")
        and "paper-only" in read("MATERIALS.md"),
        "Participation parity or safety boundary missing",
    )
    print("PASS ten 25-minute scripts, 30 routes, 20 worked swaps and 20 bridges")


def checks() -> None:
    public = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    routine = read("PRACTICE-SWAPS.md")
    need(
        re.findall(r"^## Check ([MN]) · Day (\d+)\b", public, re.MULTILINE)
        == [("M", "35"), ("N", "40")],
        "Fresh public check inventory drift",
    )
    for exact in (
        "P",
        "striped token",
        "SORTING SHELF",
        "R",
        "ring mark",
        "PARTS RACK",
    ):
        need(exact in public and exact in key, f"Fresh source/key mismatch: {exact}")
    need(
        "one workable m plan" in key.lower() and "one workable n plan" in key.lower(),
        "Separate worked guidance absent",
    )
    need(
        "one workable" not in public.lower()
        and "P and striped token appear" not in public,
        "Worked answer leaked to public check",
    )
    need(
        "striped token" not in routine and "PARTS RACK" not in routine,
        "Held source leaked into routine",
    )
    need(
        "After Check M" in table_row("PRACTICE-SWAPS.md", 35, 2)[0]
        and "After Check N" in table_row("PRACTICE-SWAPS.md", 40, 2)[0],
        "After-check practice timing absent",
    )
    need(
        "not secure exams" in public and "NOT OBSERVED" in key,
        "Public/conditional evidence boundary missing",
    )
    print("PASS fresh Check M/N sources, public learner copy and separate worked key")


def assets() -> None:
    generator = ast.parse(read("generate_assets.py"))
    page_dict = next(
        node.value
        for node in generator.body
        if isinstance(node, ast.Assign)
        and any(
            isinstance(target, ast.Name) and target.id == "PAGES"
            for target in node.targets
        )
    )
    need(
        isinstance(page_dict, ast.Dict)
        and {key.value for key in page_dict.keys if isinstance(key, ast.Constant)}
        == {*(f"print/{stem}" for stem in AIDS), *(f"check/{stem}" for stem in CHECKS)},
        "Nine-page original generator inventory drift",
    )
    alt = read("print/TEXT-ALTERNATIVES.md")
    ns = {"svg": "http://www.w3.org/2000/svg"}
    for folder, stems in (("print", AIDS), ("check", CHECKS)):
        for stem in stems:
            name = f"{folder}/{stem}"
            svg = ET.parse(ROOT / f"{name}.svg").getroot()
            need(
                svg.get("width") == "210mm"
                and svg.get("height") == "297mm"
                and svg.get("viewBox") == "0 0 794 1123"
                and svg.find("svg:title", ns) is not None
                and svg.find("svg:desc", ns) is not None,
                f"A4 SVG/accessible naming drift: {name}",
            )
            need(
                len(svg.findall(".//svg:rect", ns)) >= 4
                and len(svg.findall(".//svg:line", ns)) >= 4,
                f"Pictorial geometry drift: {name}",
            )
            printed = [node.text or "" for node in svg.findall(".//svg:text", ns)]
            for value in printed:
                need(value in alt, f"Exact text alternative omits {name}: {value}")
            need(
                f"`{name}.svg` and `{name}.pdf`" in alt
                and "**Tactile route:**" in alt
                and "**No-print route:**" in alt,
                f"Tactile/no-print alternative missing: {name}",
            )
            pdf = ROOT / f"{name}.pdf"
            info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
            pdf_text = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
            fonts = subprocess.check_output(["pdffonts", str(pdf)], text=True)
            need(
                re.search(r"Pages:\s+1\b", info)
                and "(A4)" in info
                and "FOUNDATION DRAMA" in pdf_text
                and len(pdf_text.split()) >= 25
                and "DejaVu" in fonts,
                f"Single A4 searchable PDF/font drift: {name}",
            )
    need("Bitstream" in read("print/dejavu-font-copyright.txt"), "Font rights missing")
    for phrase in ("No solved second scene", "PARTS RACK", "SORTING SHELF"):
        need(
            phrase.lower() in alt.lower(), f"Fresh print alternative missing: {phrase}"
        )
    print(
        "PASS nine illustrated original A4 SVG/PDF pairs, exact text/tactile/no-print routes and font rights"
    )


def slug(title: str) -> str:
    title = re.sub(r"\[[^]]+\]\([^)]+\)", "", title)
    title = re.sub(r"[*_]", "", title).strip().lower()
    title = re.sub(r"[^\w -]", "", title)
    return title.replace(" ", "-")


def links(*, bootstrap: bool) -> None:
    count = 0
    for path in ROOT.rglob("*.md"):
        body = path.read_text(encoding="utf-8")
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", body):
            if url.startswith(("https://", "http://", "mailto:")):
                continue
            name, marker, anchor = unquote(url).partition("#")
            target = (path.parent / name).resolve() if name else path
            if bootstrap and target == ROOT / "ASSET-MANIFEST.json":
                count += 1
                continue
            need(
                target.is_file(),
                f"Broken local link: {path.relative_to(ROOT)} -> {url}",
            )
            if marker and target.suffix.lower() == ".md":
                headings = [
                    slug(heading)
                    for heading in re.findall(
                        r"^#{1,6} (.+)$",
                        target.read_text(encoding="utf-8"),
                        re.MULTILINE,
                    )
                ]
                need(anchor in headings, f"Broken Markdown heading: {url}")
            count += 1
    print(f"PASS {count} local file/heading links")


def manifest_data() -> dict:
    files = sorted(
        path.relative_to(ROOT).as_posix()
        for path in ROOT.rglob("*")
        if path.is_file()
        and path.name != "ASSET-MANIFEST.json"
        and "__pycache__" not in path.parts
        and ".ruff_cache" not in path.parts
    )
    need(
        set(files) == REQUIRED,
        f"Pack file set drift; missing {sorted(REQUIRED - set(files))}; extra {sorted(set(files) - REQUIRED)}",
    )
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/drama/term-1/weeks-07-08",
        "created_at": "2026-09-30",
        "review_status": "author_desk_checked_pending_educator_access_cultural_and_classroom_review",
        "curriculum_source_sha256": SOURCE_SHA,
        "curriculum_codes_conditional": ["AC9ADRFE01", "AC9ADRFP01"],
        "curriculum_codes_opportunity": ["AC9ADRFD01", "AC9ADRFC01"],
        "files": {
            name: {
                "sha256": hashlib.sha256((ROOT / name).read_bytes()).hexdigest(),
                "bytes": (ROOT / name).stat().st_size,
            }
            for name in files
        },
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument(
        "--write-manifest",
        action="store_true",
        help="intentionally freeze current reviewed files",
    )
    args = parser.parse_args()
    source()
    daily()
    checks()
    assets()
    links(bootstrap=args.write_manifest)
    expected = manifest_data()
    if args.write_manifest:
        (ROOT / "ASSET-MANIFEST.json").write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
        )
        print(f"WROTE SHA-256 receipt for {len(expected['files'])} files")
    else:
        need(
            json.loads(read("ASSET-MANIFEST.json")) == expected,
            "Missing or stale SHA-256 receipt",
        )
        print(f"PASS SHA-256 receipt for {len(expected['files'])} files")
    print("PACK PASS · author desk QA only; no child, educator or funder pilot")


if __name__ == "__main__":
    main()
