#!/usr/bin/env python3
"""Read-only-by-default source, content, accessibility and hash receipt."""

from __future__ import annotations

import argparse
import ast
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
ROWS = {"AC9TDIFK01": 20098, "AC9TDIFK02": 20105, "AC9TDIFP01": 20111}
STEMS = (
    "badge-source",
    "shape-rule",
    "border-rule",
    "three-form-display",
    "trial-record",
    "source-privacy-mat",
    "check-a-display-depot",
    "check-b-tool-trial",
)
SOURCE_PHRASES = {
    "badge-source": ("CIRCLE", "TRIANGLE", "SOLID", "DASHED", "NOT SHOWN", "CODE C?"),
    "shape-rule": ("CIRCLE", "TRIANGLE", "PLACE LARGE TOKENS HERE"),
    "border-rule": ("SOLID", "DASHED", "CANNOT TELL", "PLACE TOKENS HERE"),
    "three-form-display": ("OBJECT", "PICTURE", "SYMBOL", "CS TD CD TS C?"),
    "trial-record": ("SAMPLE P", "SAMPLE Q", "SAMPLE R", "SAMPLE S", "NOT TESTED"),
    "source-privacy-mat": (
        "MY NAME",
        "MY FACE PHOTO",
        "MY VOICE RECORDING",
        "ACTUAL DEVICE ACTION",
    ),
    "check-a-display-depot": ("SQUARE", "OVAL", "DOTTED", "PLAIN", "NOT SHOWN"),
    "check-b-tool-trial": (
        "SAMPLE K",
        "SAMPLE L",
        "SAMPLE M",
        "SAMPLE N",
        "NOT TESTED",
    ),
}


def need(ok: object, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def read(name: str) -> str:
    return (ROOT / name).read_text(encoding="utf-8")


def source() -> None:
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    imported = json.loads(
        (STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8")
    )
    snapshot = json.loads(read("source-snapshot.json"))
    crosswalk = read("CURRICULUM-CROSSWALK.md")
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == imported["source_sha256"]
        == snapshot["workbook_sha256"]
        == SOURCE_SHA,
        "Pinned official workbook/import/SHA drift",
    )
    need(snapshot["content_rows"] == ROWS, "Official source row drift")
    need(
        snapshot["queensland_prep_alignment_url"] in crosswalk
        and snapshot["queensland_prep_entry_point_url"] in crosswalk
        and snapshot["official_workbook_url"] in crosswalk,
        "Official URL pin missing",
    )
    records = {
        row["code"]: row
        for row in imported["records"]
        if row.get("record_type") == "content_description" and row.get("code") in ROWS
    }
    need(set(records) == set(ROWS), "Official content description missing")
    for code, number in ROWS.items():
        record = records[code]
        attrs = record["attributes"]
        need(
            record["source_row"] == number
            and attrs["level"] == "Foundation Year"
            and attrs["learning_area"] == "Technologies"
            and attrs["subject"] == "Digital Technologies",
            f"Official source placement drift: {code}",
        )
        expected = (
            rf"^\| {code} \| {number} \| Technologies · Digital Technologies · Foundation Year \| "
            rf"{re.escape(record['plain_text'])} \|"
        )
        need(
            re.search(expected, crosswalk, re.MULTILINE),
            f"Exact wording missing: {code}",
        )
    need(
        "AC9TDI2K02" in crosswalk
        and "NOT OBSERVED" in crosswalk
        and "fictional" in crosswalk.lower(),
        "QCAA typo, conditional gate or fictional-source limit missing",
    )
    print("PASS pinned official ACARA workbook, 3 exact rows and QCAA Prep source pins")


def daily() -> None:
    lessons = read("LESSONS.md")
    matches = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", lessons, re.MULTILINE))
    need(
        [int(mark.group(2)) for mark in matches] == list(range(21, 31)),
        "Daily sequence incomplete",
    )
    for index, mark in enumerate(matches):
        day = int(mark.group(2))
        week = int(mark.group(1))
        end = matches[index + 1].start() if index + 1 < len(matches) else len(lessons)
        body = lessons[mark.end() : end]
        times = [
            int(value)
            for value in re.findall(
                r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*", body, re.MULTILINE
            )
        ]
        need(
            week == (5 if day <= 25 else 6) and times == [3, 4, 5, 7, 4, 2],
            f"Day {day} schedule drift: {times}",
        )
        target = re.search(r"^\*\*Target/codes:\*\* ([^\n]+)", body, re.MULTILINE)
        need(
            target and "AC9TDIFK02" in target.group(1), f"Day {day} target/code missing"
        )
        need(
            ("**Notice · 4 min.**" in body or "**Self-check · 4 min.**" in body)
            and ("save" in body.lower() or "record" in body.lower()),
            f"Day {day} observation and response absent",
        )
        if day == 29:
            need(
                "AC9TDIFK01 conditional" in target.group(1),
                "Actual-system code must be conditional",
            )
        elif day != 29:
            need(
                "AC9TDIFK01" not in target.group(1),
                f"Day {day} claims system code without device",
            )
    cards = read("LEARNER-CARDS.md")
    swaps = read("PRACTICE-SWAPS.md")
    bridges = read("EXAMPLE-BANK.md")
    prompts = read("PROMPTS.md")
    for name, body in (
        ("learner routes", cards),
        ("worked swaps", swaps),
        ("interest bridges", bridges),
        ("prompts", prompts),
    ):
        ids = [
            int(value)
            for value in re.findall(r"^\| (2[1-9]|30) \|", body, re.MULTILINE)
        ]
        need(ids == list(range(21, 31)), f"Ten daily rows missing in {name}")
    for day in range(21, 31):
        row = re.search(rf"^\| {day} \| (.+) \| (.+) \| (.+) \|$", cards, re.MULTILINE)
        need(
            row and all(len(cell) > 35 for cell in row.groups()),
            f"Day {day} three substantive routes absent",
        )
        for name, body in (("worked swaps", swaps), ("interest bridges", bridges)):
            row = re.search(rf"^\| {day} \| (.+) \| (.+) \|$", body, re.MULTILINE)
            need(
                row and all(len(cell) > 35 for cell in row.groups()),
                f"Day {day} two {name} absent",
            )
    need(
        "not fixed learning styles" in cards and "NOT OBSERVED" in cards,
        "Route and device limits missing",
    )
    need(
        "**After Check:**" in swaps and "fictional" in lessons.lower(),
        "Fresh-after-check or fictional boundary missing",
    )
    print(
        "PASS ten 25-minute scripts, 30 same-target routes, 20 worked swaps, 20 bridges"
    )


def checks() -> None:
    public = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    routine = read("PRACTICE-SWAPS.md")
    need(
        re.findall(r"^## Check ([AB]) · Day (\d+)\b", public, re.MULTILINE)
        == [("A", "25"), ("B", "30")],
        "Fresh check inventory/day drift",
    )
    for exact in ("QD", "OD", "QP", "O?", "K", "L", "M", "N"):
        need(exact in key, f"Check key omits symbol/ID: {exact}")
    for exact in (
        "square, dotted",
        "oval, dotted",
        "square, plain",
        "oval, border NOT SHOWN",
        "K SLIPPED",
        "L NOT TESTED",
        "M HELD",
        "N HELD",
    ):
        need(exact in public, f"Public fresh source drift: {exact}")
    need("SQUARE 1, 3, 5" in key and "OVAL 2, 4" in key, "Check A shape key drift")
    need(
        "DOTTED 1, 2" in key and "PLAIN 3, 5" in key and "CANNOT TELL 4" in key,
        "Check A border key drift",
    )
    need(
        "HELD M, N" in key and "SLIPPED K" in key and "NOT TESTED L" in key,
        "Check B outcome key drift",
    )
    need(
        "1, 3, 5" not in public and "HELD M, N" not in public,
        "Solved grouping leaked to public checks",
    )
    need(
        "square-dotted" not in routine and "K SLIPPED" not in routine,
        "Fresh source leaked into routine",
    )
    need(
        "formative" in public.lower()
        and "public" in public.lower()
        and "first response" in key.lower(),
        "Assessment limit or first-work handling missing",
    )
    print(
        "PASS two fresh public formative checks and separate source-backed teacher key"
    )


def assets() -> None:
    generator = ast.parse(read("print/generate_print.py"))
    pages = next(
        node.value
        for node in generator.body
        if isinstance(node, ast.Assign)
        and any(
            isinstance(target, ast.Name) and target.id == "PAGES"
            for target in node.targets
        )
    )
    need(
        isinstance(pages, ast.Dict)
        and {key.value for key in pages.keys if isinstance(key, ast.Constant)}
        == set(STEMS),
        "Eight-page generator inventory drift",
    )
    alternatives = read("print/TEXT-ALTERNATIVES.md")
    ns = {"svg": "http://www.w3.org/2000/svg"}
    for stem in STEMS:
        svg = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
        need(
            svg.get("width") == "210mm"
            and svg.get("height") == "297mm"
            and svg.find("svg:title", ns) is not None
            and svg.find("svg:desc", ns) is not None,
            f"SVG A4 or alternative description drift: {stem}",
        )
        printed = [node.text or "" for node in svg.findall(".//svg:text", ns)]
        for value in printed:
            need(
                value in alternatives,
                f"SVG printed text absent from exact alternative: {stem}: {value}",
            )
        need(
            len(svg.findall(".//svg:rect", ns)) >= 3
            and f"{stem}.svg / {stem}.pdf" in alternatives
            and "Tactile alternative" in alternatives,
            f"Original geometry or tactile route missing: {stem}",
        )
        if stem in ("badge-source", "shape-rule", "check-a-display-depot"):
            need(
                sum(
                    len(svg.findall(f".//svg:{tag}", ns))
                    for tag in ("circle", "polygon", "ellipse")
                )
                >= 2,
                f"Illustrated shape geometry absent: {stem}",
            )
        pdf = ROOT / "print" / f"{stem}.pdf"
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        pdf_text = " ".join(
            subprocess.check_output(["pdftotext", str(pdf), "-"], text=True).split()
        )
        need(
            re.search(r"Pages:\s+1\b", info)
            and "(A4)" in info
            and len(pdf_text.split()) >= 25,
            f"PDF page/text drift: {stem}",
        )
        for phrase in SOURCE_PHRASES[stem]:
            need(
                phrase in pdf_text and phrase in alternatives,
                f"PDF/alternative fact drift: {stem}: {phrase}",
            )
    font = read("print/DEJAVU-FONT-LICENSE.txt")
    need("DejaVu" in font and "Bitstream" in font, "Print font rights missing")
    print(
        "PASS eight original illustrated A4 SVG/PDF pairs and exact text/tactile alternatives"
    )


def interactive() -> None:
    html = read("interactive/rule-sorter.html")
    paper = read("interactive/PAPER-EQUIVALENT.md")
    need(
        html.count('class="card"') == 5
        and 'role="status" aria-live="polite"' in html
        and "document.createElement('select')" in html
        and "CANNOT TELL" in html
        and "NOT OBSERVED" not in html,
        "Interactive source, input or feedback drift",
    )
    need(
        not re.search(
            r"fetch\s*\(|XMLHttpRequest|localStorage|sessionStorage|"
            r"<form\b|<input\b|https?://|<script\s+src=|"
            r"navigator\.(?:mediaDevices|sendBeacon)|indexedDB|document\.cookie",
            html,
            re.IGNORECASE,
        ),
        "Unexpected network, storage or personal-input surface",
    )
    need(
        all(
            phrase in paper
            for phrase in ("1,3,5", "2,4", "1,4", "2,3", "AC9TDIFK02", "AC9TDIFK01")
        )
        and "NOT OBSERVED" in paper,
        "Paper parity or device evidence boundary drift",
    )
    result = subprocess.check_output(
        ["node", str(ROOT / "interactive/verify_sorter.mjs")], text=True
    )
    need(
        result.startswith("PASS fixed source; both rules"),
        "Sorter's executable interaction simulation failed",
    )
    print(
        "PASS contained offline sorter static safety, executable interaction simulation and paper route"
    )


def slug(title: str) -> str:
    title = re.sub(r"\[[^]]+\]\([^)]+\)", "", title)
    title = re.sub(r"[*_]", "", title).strip().lower()
    title = re.sub(r"[^\w -]", "", title)
    return title.replace(" ", "-")


def links(*, bootstrap: bool) -> None:
    count = 0
    for path in ROOT.rglob("*.md"):
        body = path.read_text(encoding="utf-8")
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", body):
            if url.startswith(("https://", "http://", "mailto:")):
                continue
            name, marker, anchor = unquote(url).partition("#")
            target = (path.parent / name).resolve() if name else path
            if bootstrap and target == ROOT / "manifest.json":
                count += 1
                continue
            need(
                target.is_file(),
                f"Broken local link: {path.relative_to(ROOT)} -> {url}",
            )
            if marker and target.suffix.lower() == ".md":
                headings = [
                    slug(heading)
                    for heading in re.findall(
                        r"^#{1,6} (.+)$",
                        target.read_text(encoding="utf-8"),
                        re.MULTILINE,
                    )
                ]
                need(anchor in headings, f"Broken Markdown heading: {url}")
            count += 1
    print(f"PASS {count} local file/heading links")


def manifest_data() -> dict:
    files = sorted(
        path
        for path in ROOT.rglob("*")
        if path.is_file()
        and path.name != "manifest.json"
        and "__pycache__" not in path.parts
    )
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/digital-technologies/term-1/weeks-05-06",
        "created_at": "2026-09-30",
        "review_status": "author_desk_checked_pending_educator_and_device_access_review",
        "curriculum_source_sha256": SOURCE_SHA,
        "curriculum_codes_partial": ["AC9TDIFK02", "AC9TDIFP01"],
        "curriculum_code_conditional_actual_system": "AC9TDIFK01",
        "files": {
            str(path.relative_to(ROOT)): {
                "sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
                "bytes": path.stat().st_size,
            }
            for path in files
        },
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument(
        "--write-manifest",
        action="store_true",
        help="intentionally freeze current reviewed files",
    )
    args = parser.parse_args()
    source()
    daily()
    checks()
    assets()
    interactive()
    links(bootstrap=args.write_manifest)
    expected = manifest_data()
    if args.write_manifest:
        (ROOT / "manifest.json").write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
        )
        print(f"WROTE SHA-256 receipt for {len(expected['files'])} files")
    else:
        need(
            json.loads(read("manifest.json")) == expected,
            "Missing or stale SHA-256 receipt",
        )
        print(f"PASS SHA-256 receipt for {len(expected['files'])} files")
    print("PACK PASS · author desk QA only; no educator, child or funder pilot")


if __name__ == "__main__":
    main()
