# SPDX-License-Identifier: Apache-2.0
"""Read-only by default: source, lesson, link, A4, SHA and regeneration audit."""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import shutil
import struct
import subprocess
import sys
import tempfile
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

from pypdf import PdfReader

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
ROWS = {"AC9AVAFE01": 21465, "AC9AVAFD01": 21474, "AC9AVAFC01": 21484, "AC9AVAFP01": 21493}
DESCRIPTIONS = {
    "AC9AVAFE01": "explore how and why the arts are important for people and communities",
    "AC9AVAFD01": "use play, imagination, arts knowledge, processes and/or skills to discover possibilities and develop ideas",
    "AC9AVAFC01": "create arts works that communicate ideas",
    "AC9AVAFP01": "share their arts works with audiences",
}
CONTEXT_ROWS = {'AC9TDIFK02': 20105, 'AC9AMAFD01': 20976, 'AC9AMAFC01': 20986, 'AC9AMAFP01': 20995, 'AC9HSFK03': 1221, 'AC9HSFK04': 1226, 'AC9HPFM03': 3215}
CONTEXT_DESCRIPTIONS = {'AC9HSFK03': 'the features of familiar places they belong to, why some places are special and how places can be looked after', 'AC9HSFK04': 'the importance of Country/Place to First Nations Australians and the Country/Place on which the school is located', 'AC9HPFM03': 'participate in a range of activities in natural and outdoor settings and explore the benefits of being physically active', 'AC9TDIFK02': 'represent data as objects, pictures and symbols', 'AC9AMAFD01': 'use play, imagination, arts knowledge, processes and/or skills to discover possibilities and develop ideas', 'AC9AMAFC01': 'create arts works that communicate ideas', 'AC9AMAFP01': 'share their arts works with audiences'}

URLS = (
    "https://www.qcaa.qld.edu.au/p-10/aciq/version-9/learning-areas/p-10-the-arts/visual-arts",
    "https://www.qcaa.qld.edu.au/downloads/aciqv9/the-arts/curriculum/ac9_visual_arts_prep_as_cd.alignment.pdf",
    "https://www.qcaa.qld.edu.au/downloads/aciqv9/the-arts/assessment/ac9_arts_non_perform_tc_prep.pdf",
)
STEMS = ('rubbing-workshop', 'master-plan', 'impression-choice', 'place-source', 'art-canvas', 'making-record', 'check-y-bands', 'check-z-shelf')
PDF_PHRASES = {'rubbing-workshop': ['RESULT UNKNOWN', 'Crayon marks the TOP paper', 'not a photograph'], 'master-plan': ['MY MASTER PLAN', 'BAR', 'DISC', 'ARCH'], 'impression-choice': ['FICTIONAL DRAWN PAPER', 'NOT AN ACTUAL RUBBING', 'MY SELECTED PATCHES'], 'place-source': ['COURTYARD C', 'middle path', 'permission shown', 'MY INVENTED ART CHOICE'], 'art-canvas': ['MY ORIGINAL ART', 'KEEP', 'POSTCARD', 'PLAN ONLY'], 'making-record': ['MASTER', 'RUBBING TRIED', 'NOT TRIED', 'ARRANGED', 'NOT OBSERVED'], 'check-y-bands': ['FRESH Y', 'OVAL upper left', 'lower-right band slightly higher', 'MY OWN MASTER', 'NO RUBBING RESULT'], 'check-z-shelf': ['FRESH Z', 'EMPTY TABLE', 'TRAYS', 'No people', 'MY SOURCE FEATURE', 'Whose agreement']}

def need(ok: object, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def read(name: str) -> str:
    return (ROOT / name).read_text(encoding="utf-8")


def source() -> None:
    workbook = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    imported = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8"))
    snapshot = json.loads(read("source-snapshot.json"))
    crosswalk = read("CURRICULUM-CROSSWALK.md")
    rights = read("SOURCE-AND-RIGHTS.md")
    need(hashlib.sha256(workbook.read_bytes()).hexdigest() == imported["source_sha256"] == snapshot["workbook_sha256"] == SOURCE_SHA, "Official workbook/import/snapshot SHA drift")
    need(snapshot["content_rows"] == ROWS and snapshot["contextual_rows"] == CONTEXT_ROWS, "Exact source row drift")
    need([snapshot[key] for key in ("queensland_prep_entry_point_url", "queensland_prep_alignment_url", "queensland_prep_techniques_url")] == list(URLS), "QCAA source URL drift")
    need(all(url in crosswalk and url in rights for url in URLS), "QCAA primary link missing")
    records = {item["code"]: item for item in imported["records"] if item.get("code") in set(ROWS) | set(CONTEXT_ROWS)}
    for code, row in ROWS.items():
        item = records[code]
        need(item["source_row"] == row and item["attributes"]["level"] == "Foundation Year" and item["attributes"]["subject"] == "Visual Arts", f"Source pin mismatch {code}")
        need(item["plain_text"] == DESCRIPTIONS[code] and DESCRIPTIONS[code] in crosswalk, f"Description mismatch {code}")
    for code, row in CONTEXT_ROWS.items():
        item = records[code]
        need(item["source_row"] == row and item["plain_text"] == CONTEXT_DESCRIPTIONS[code] and CONTEXT_DESCRIPTIONS[code] in crosswalk, f"Context source mismatch {code}")
    need("403" in rights and "not observed" in crosswalk.lower(), "Official fetch or claim boundary missing")
    receipts = json.loads(read("source-fetch-receipts.json"))["receipts"]
    need(len(receipts) == 4 and receipts[0]["http_status"] == 200 and receipts[0]["sha256"] == SOURCE_SHA, "Fresh ACARA fetch receipt drift")
    need(all(item["http_status"] == 403 for item in receipts[1:]), "QCAA blocked-fetch receipt drift")
    need("body unverified" in snapshot["qcaa_evidence_status"], "Current QCAA PDF limitation missing")
    week15 = STUDIO / "content/foundation/term-2/week-17-what-a-picture-shows.md"
    week16 = STUDIO / "content/foundation/term-2/week-18-care-for-a-shared-place.md"
    need("frame" in week15.read_text() and "permission" in week16.read_text(), "Integrated context drift")
    print("PASS official ACARA SHA, exact rows/descriptions and QCAA pins with honest fetch limit")


def pedagogy() -> None:
    lessons = read("LESSONS.md")
    blocks = re.split(r"(?=^## Week \d+ · Day \d+ ·)", lessons, flags=re.MULTILINE)[1:]
    days = []
    for block in blocks:
        day = int(re.search(r"^## Week \d+ · Day (\d+) ·", block).group(1))
        days.append(day)
        minutes = [int(x) for x in re.findall(r"^\d\. \*\*[^*]+ · (\d+) min\.\*\*", block, flags=re.MULTILINE)]
        need(minutes == [3, 4, 5, 7, 4, 2] and sum(minutes) == 25, f"Day {day} timing drift: {minutes}")
        need("**Goal/codes:**" in block and "**Prepare:**" in block, f"Day {day} runnable fields missing")
        need("AC9AVAFD01" in block or day == 89, f"Day {day} Visual Arts goal missing")
    need(days == list(range(81, 91)), f"Expected ten daily scripts: {days}")
    cards = read("LEARNER-CARDS.md")
    for day in days:
        segment = re.search(rf"^## Day {day} .*?(?=^## Day \d+ |^## Record|\Z)", cards, re.MULTILINE | re.DOTALL)
        need(segment and all(re.search(rf"\*\*{route}(?: |:)", segment.group()) for route in "ABC") and "**At home" in segment.group(), f"Day {day} route/home missing")
    swaps = read("PRACTICE-SWAPS.md")
    need([int(x) for x in re.findall(r"^\| (\d\d) \|", swaps, flags=re.MULTILINE)] == days, "Twenty domain swaps absent")
    checks = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    need("## Check Y" in checks and "## Check Z" in checks and "## Check Y" in key and "## Check Z" in key, "Fresh checks/separate key missing")
    need("Possible response" not in checks and "Possible" in key, "Learner copy leaks worked solution")
    need("PLAN ONLY" in checks and "not secure" in checks and "term grade" in key, "Assessment boundary missing")
    need(sum(line.count("AFTER first check capture only") for line in swaps.splitlines() if line.startswith("| ")) == 4, "Check-day worked models must follow first capture")
    need("## Parent/carer pass" in read("RUN-THROUGH.md"), "Parent desk pass missing")
    need("outdoor participation" in checks and "Country/Place" in checks and "actual rubbing" in checks, "Cross-subject evidence boundary missing")
    print("PASS ten timed 25-minute scripts, 30 routes, 20 worked swaps and two held checks")


def assets() -> None:
    alternatives = read("print/TEXT-ALTERNATIVES.md")
    need(len(STEMS) == 8 and len(list((ROOT / "print").glob("*.svg"))) == 8 and len(list((ROOT / "print").glob("*.pdf"))) == 8, "Eight pairs missing or extra")
    for stem in STEMS:
        svg = ROOT / "print" / f"{stem}.svg"
        pdf = ROOT / "print" / f"{stem}.pdf"
        xml = ET.parse(svg).getroot()
        need(xml.attrib.get("width") == "210mm" and xml.attrib.get("height") == "297mm" and xml.attrib.get("viewBox") == "0 0 794 1123", f"A4 SVG geometry mismatch: {stem}")
        need(xml.find("{http://www.w3.org/2000/svg}title") is not None and xml.find("{http://www.w3.org/2000/svg}desc") is not None, f"SVG text alternative missing: {stem}")
        pages = PdfReader(pdf).pages
        need(len(pages) == 1 and abs(float(pages[0].mediabox.width) - 595.28) < 1 and abs(float(pages[0].mediabox.height) - 841.89) < 1, f"A4 PDF mismatch: {stem}")
        extracted = pages[0].extract_text()
        compact_text = re.sub(r"\s+", "", extracted)
        need(all(re.sub(r"\s+", "", phrase) in compact_text for phrase in PDF_PHRASES[stem]), f"PDF text mismatch: {stem}")
        bbox = subprocess.run(["pdftotext", "-bbox", str(pdf), "-"], check=True, capture_output=True, text=True).stdout
        bbox_root = ET.fromstring(bbox)
        for item in bbox_root.iter():
            if item.tag.endswith("}word"):
                need(0 <= float(item.attrib["xMin"]) < float(item.attrib["xMax"]) <= 595.3 and 0 <= float(item.attrib["yMin"]) < float(item.attrib["yMax"]) <= 841.9, f"Clipped PDF text: {stem}")
        need(f"({stem}.pdf)" in alternatives and f"({stem}.svg)" in alternatives, f"Alternative pair link missing: {stem}")
    need("not tagged" in alternatives.lower() and "tactile" in alternatives.lower() and "no print/touch" in alternatives.lower(), "Linear/tactile/no-print boundary absent")
    namespace = {"s": "http://www.w3.org/2000/svg"}
    for stem in STEMS:
        xml = ET.parse(ROOT / "print" / (stem + ".svg")).getroot()
        section = alternatives.split("## " + stem + "\n", 1)[1].split("\n## ", 1)[0]
        exact = section.split("**Exact visible wording in reading order:**", 1)[1]
        words = [node.text or "" for node in xml.findall("s:text", namespace)]
        need([line[2:] for line in exact.splitlines() if line.startswith("- ")] == words, "Exact alternative wording/order drift: " + stem)
    y = ET.parse(ROOT / "print/check-y-bands.svg").getroot()
    source_oval = [n for n in y.findall("s:ellipse", namespace) if n.get("data-part", "").startswith("Y broad")]
    bands = [n for n in y.findall("s:rect", namespace) if n.get("data-part", "").startswith("Y band")]
    need(len(source_oval) == 1 and len(bands) == 2, "Y source must keep one oval/two bands")
    need((source_oval[0].get("cx"), source_oval[0].get("cy")) == ("223", "282"), "Y oval source position drift")
    need([(n.get("x"), n.get("y"), n.get("width"), n.get("height")) for n in bands] == [("146","428","172","38"),("450","392","176","38")], "Y offset bands source drift")
    need(len(y.findall("s:path", namespace)) == 1, "Y must not contain worked rubbing marks")
    z = ET.parse(ROOT / "print/check-z-shelf.svg").getroot()
    source = {n.get("data-part"): n for n in z.findall("s:rect", namespace) if n.get("data-part", "").startswith("Z ")}
    need(all(n in source for n in ("Z shelf left","Z empty table middle","Z upper tray right","Z lower tray right")), "Z supplied features missing")
    need([source[n].get("x") for n in ("Z shelf left","Z empty table middle","Z upper tray right")] == ["91","293","590"], "Z source left/middle/right drift")
    need(source["Z upper tray right"].get("y") == "262" and source["Z lower tray right"].get("y") == "375", "Z vertical tray order drift")
    shelf_lines = [n for n in z.findall("s:path", namespace) if n.get("data-part") == "Z two horizontal shelves"]
    need(len(shelf_lines) == 1 and shelf_lines[0].get("d") == "M94 304 L188 304 M94 379 L188 379", "Z shelf source shape drift")
    need(len(z.findall("s:path", namespace)) == 3, "Z must not contain an adult care-art answer")
    print("PASS eight A4 pairs, exact word/layout alternatives, fresh Y/Z source geometry and blank answer boundaries")


def regeneration() -> None:
    with tempfile.TemporaryDirectory(prefix="subjectnest-va-w17-18-") as tmp:
        temp = Path(tmp)
        shutil.copy2(ROOT / "print/generate_print.py", temp / "generate_print.py")
        subprocess.run([sys.executable, str(temp / "generate_print.py")], check=True, capture_output=True)
        for stem in STEMS:
            for ext in ("svg", "pdf"):
                name = f"{stem}.{ext}"
                need(hashlib.sha256((temp / name).read_bytes()).digest() == hashlib.sha256((ROOT / "print" / name).read_bytes()).digest(), f"Nondeterministic or stale asset: {name}")
    print("PASS deterministic regeneration of all sixteen A4 assets")


def digital() -> None:
    html = read("digital/index.html")
    js = read("digital/app.js")
    css = read("digital/style.css")
    guide = read("digital/README.md")
    need("Content-Security-Policy" in html and "default-src 'none'" in html and '<script src="app.js" defer>' in html and '<link rel="stylesheet" href="style.css">' in html, "Offline browser scaffold drift")
    need(all(f'id="{name}"' in html for name in ("arrangement", "front", "use", "picture", "reset", "description", "status")), "Browser controls missing")
    need("aria-live=\"polite\"" in html and ":focus-visible" in css and "addEventListener" in js and "textContent" in js, "Browser access or safe text route missing")
    need(not re.search(r"\b(?:fetch|XMLHttpRequest|localStorage|sessionStorage|sendBeacon|WebSocket|serviceWorker)\b", js), "Unexpected browser network/storage API")
    need("closing or reloading clears" in (html + js).lower() and "screen" in read("MATERIALS.md").lower(), "Privacy or no-screen boundary missing")
    subprocess.run(["node", "--check", str(ROOT / "digital/app.js")], check=True, capture_output=True)
    for name, width in (("desktop", 1280), ("mobile", 390)):
        preview = ROOT / "digital/previews" / f"{name}.png"
        raw = preview.read_bytes()
        need(raw[:8] == b"\x89PNG\r\n\x1a\n" and struct.unpack(">I", raw[16:20])[0] == width and f"previews/{name}.png" in guide, f"Browser preview missing: {name}")
    need("1280, 390 and 320" in guide and (ROOT / "digital/qa_browser.cjs").is_file(), "Substantive browser QA route missing")
    print("PASS offline text/keyboard collage planner, JS syntax and two preview dimensions")


def slug(title: str) -> str:
    title = re.sub(r"\[([^]]+)\]\([^)]+\)", r"\1", title)
    title = re.sub(r"[*_]", "", title).strip().lower()
    return re.sub(r"[^\w -]", "", title).replace(" ", "-")


def links(*, bootstrap: bool) -> None:
    count = 0
    for path in ROOT.rglob("*.md"):
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("https://", "http://", "mailto:")):
                continue
            name, marker, anchor = unquote(url).partition("#")
            target = (path.parent / name).resolve() if name else path
            if bootstrap and target == ROOT / "manifest.json":
                count += 1
                continue
            need(target.is_file(), f"Broken local link: {path.relative_to(ROOT)} -> {url}")
            if marker and target.suffix == ".md":
                headings = [slug(value) for value in re.findall(r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)]
                need(anchor in headings, f"Broken heading anchor: {url}")
            count += 1
    print(f"PASS {count} local links and heading anchors")


def manifest_data() -> dict:
    files = sorted(path for path in ROOT.rglob("*") if path.is_file() and path.name != "manifest.json" and "__pycache__" not in path.parts)
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/visual-arts/term-2/weeks-17-18",
        "created_at": "2026-09-30",
        "pack_version": "0.1.0-draft",
        "review_status": "author_desk_checked_pending_school_child_accessibility_and_cultural_review",
        "curriculum_source_sha256": SOURCE_SHA,
        "curriculum_codes_partial_conditional": sorted(ROWS),
        "contextual_codes_separate_evidence": sorted(CONTEXT_ROWS),
        "files": {str(path.relative_to(ROOT)): {"sha256": hashlib.sha256(path.read_bytes()).hexdigest(), "bytes": path.stat().st_size} for path in files},
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true", help="One-time authoring update; default is read-only")
    args = parser.parse_args()
    source()
    pedagogy()
    assets()
    regeneration()
    digital()
    links(bootstrap=args.write_manifest)
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
        print(f"WROTE SHA-256 manifest for {len(expected['files'])} files")
    else:
        need(json.loads(manifest.read_text(encoding="utf-8")) == expected, "SHA-256 manifest missing or stale")
        print(f"PASS SHA-256 manifest for {len(expected['files'])} files")
    print("PACK PASS · author desk QA only; no child or classroom trial")


if __name__ == "__main__":
    main()
