# SPDX-License-Identifier: Apache-2.0
"""Read-only by default: source, lesson, link, A4, SHA and regeneration audit."""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import shutil
import struct
import subprocess
import sys
import tempfile
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

from pypdf import PdfReader

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
ROWS = {"AC9AVAFE01": 21465, "AC9AVAFD01": 21474, "AC9AVAFC01": 21484, "AC9AVAFP01": 21493}
DESCRIPTIONS = {
    "AC9AVAFE01": "explore how and why the arts are important for people and communities",
    "AC9AVAFD01": "use play, imagination, arts knowledge, processes and/or skills to discover possibilities and develop ideas",
    "AC9AVAFC01": "create arts works that communicate ideas",
    "AC9AVAFP01": "share their arts works with audiences",
}
CONTEXT_ROWS = {"AC9HSFS05": 1251, "AC9HSFK01": 1213, "AC9SFI03": 18562, "AC9SFU01": 18517}
CONTEXT_DESCRIPTIONS = {
    "AC9HSFS05": "share narratives and observations, using sources and terms about the past and places",
    "AC9HSFK01": "the people in their family, where they were born and raised, and how they are related to each other",
    "AC9SFI03": "represent observations in provided templates and identify patterns with guidance",
    "AC9SFU01": "observe external features of plants and animals and describe ways they can be grouped based on these features",
}

URLS = (
    "https://www.qcaa.qld.edu.au/p-10/aciq/version-9/learning-areas/p-10-the-arts/visual-arts",
    "https://www.qcaa.qld.edu.au/downloads/aciqv9/the-arts/curriculum/ac9_visual_arts_prep_as_cd.alignment.pdf",
    "https://www.qcaa.qld.edu.au/downloads/aciqv9/the-arts/assessment/ac9_arts_non_perform_tc_prep.pdf",
)
STEMS = (
    "source-and-choice", "short-long-marks", "close-spaced-marks", "illustration-canvas",
    "plant-view-note", "art-source-record", "check-u-lantern", "check-v-edge",
)
PDF_PHRASES = {
    "source-and-choice": ("STORY SAYS", "Sol made a paper flag.", "surface marks are NOT stated", "ADULT EXAMPLE"),
    "short-long-marks": ("SHORT MARKS", "LONG MARKS", "real surface feels"),
    "close-spaced-marks": ("SAME FOUR BROAD DASHES", "CLOSE TOGETHER", "SPACED APART"),
    "illustration-canvas": ("MY ORIGINAL ILLUSTRATION", "The source says", "PLAN ONLY"),
    "plant-view-note": ("FICTIONAL PAPER PLANT PRACTICE", "ACTUAL APPROVED VIEW / FICTION", "invented growth"),
    "art-source-record": ("SOURCE / DATE / VIEW", "MY ART MARK", "ACTUAL MARKER / SUPPORT", "ACTUAL WILLING VIEWER"),
    "check-u-lantern": ("FICTIONAL CARD LANTERN", "ONE WIDE RECTANGULAR OPENING", "IDEA 1", "IDEA 2"),
    "check-v-edge": ("FICTIONAL PAPER GARDEN CARDS", "STORY MONDAY", "STORY THURSDAY", "SAME three-bump paper edge"),
}


def need(ok: object, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def read(name: str) -> str:
    return (ROOT / name).read_text(encoding="utf-8")


def source() -> None:
    workbook = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    imported = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8"))
    snapshot = json.loads(read("source-snapshot.json"))
    crosswalk = read("CURRICULUM-CROSSWALK.md")
    rights = read("SOURCE-AND-RIGHTS.md")
    need(hashlib.sha256(workbook.read_bytes()).hexdigest() == imported["source_sha256"] == snapshot["workbook_sha256"] == SOURCE_SHA, "Official workbook/import/snapshot SHA drift")
    need(snapshot["content_rows"] == ROWS and snapshot["contextual_rows"] == CONTEXT_ROWS, "Exact source row drift")
    need([snapshot[key] for key in ("queensland_prep_entry_point_url", "queensland_prep_alignment_url", "queensland_prep_techniques_url")] == list(URLS), "QCAA source URL drift")
    need(all(url in crosswalk and url in rights for url in URLS), "QCAA primary link missing")
    records = {item["code"]: item for item in imported["records"] if item.get("code") in set(ROWS) | set(CONTEXT_ROWS)}
    for code, row in ROWS.items():
        item = records[code]
        need(item["source_row"] == row and item["attributes"]["level"] == "Foundation Year" and item["attributes"]["subject"] == "Visual Arts", f"Source pin mismatch {code}")
        need(item["plain_text"] == DESCRIPTIONS[code] and DESCRIPTIONS[code] in crosswalk, f"Description mismatch {code}")
    for code, row in CONTEXT_ROWS.items():
        item = records[code]
        need(item["source_row"] == row and item["plain_text"] == CONTEXT_DESCRIPTIONS[code] and CONTEXT_DESCRIPTIONS[code] in crosswalk, f"Context source mismatch {code}")
    need("403" in rights and "not observed" in crosswalk.lower(), "Official fetch or claim boundary missing")
    receipts = json.loads(read("source-fetch-receipts.json"))["receipts"]
    need(len(receipts) == 4 and receipts[0]["http_status"] == 200 and receipts[0]["sha256"] == SOURCE_SHA, "Fresh ACARA fetch receipt drift")
    need(all(item["http_status"] == 403 for item in receipts[1:]), "QCAA blocked-fetch receipt drift")
    need("body unverified" in snapshot["qcaa_evidence_status"], "Current QCAA PDF limitation missing")
    integrated = STUDIO / "content/foundation/term-2/week-13-people-in-a-story.md"
    need(all(phrase in integrated.read_text(encoding="utf-8") for phrase in ("cardboard boats", "fold the front", "Sol made a paper flag")), "Boat illustration source drift")
    print("PASS official ACARA SHA, exact rows/descriptions and QCAA pins with honest fetch limit")


def pedagogy() -> None:
    lessons = read("LESSONS.md")
    blocks = re.split(r"(?=^## Week \d+ · Day \d+ ·)", lessons, flags=re.MULTILINE)[1:]
    days = []
    for block in blocks:
        day = int(re.search(r"^## Week \d+ · Day (\d+) ·", block).group(1))
        days.append(day)
        minutes = [int(x) for x in re.findall(r"^\d\. \*\*[^*]+ · (\d+) min\.\*\*", block, flags=re.MULTILINE)]
        need(minutes == [3, 4, 5, 7, 4, 2] and sum(minutes) == 25, f"Day {day} timing drift: {minutes}")
        need("**Goal/codes:**" in block and "**Prepare:**" in block, f"Day {day} runnable fields missing")
        need("AC9AVAFD01" in block or day == 69, f"Day {day} Visual Arts goal missing")
    need(days == list(range(61, 71)), f"Expected ten daily scripts: {days}")
    cards = read("LEARNER-CARDS.md")
    for day in days:
        segment = re.search(rf"^## Day {day} .*?(?=^## Day \d+ |^## Record|\Z)", cards, re.MULTILINE | re.DOTALL)
        need(segment and all(re.search(rf"\*\*{route}(?: |:)", segment.group()) for route in "ABC") and "**At home" in segment.group(), f"Day {day} route/home missing")
    swaps = read("PRACTICE-SWAPS.md")
    need([int(x) for x in re.findall(r"^\| (\d\d) \|", swaps, flags=re.MULTILINE)] == days, "Twenty domain swaps absent")
    checks = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    need("## Check U" in checks and "## Check V" in checks and "## Check U" in key and "## Check V" in key, "Fresh checks/separate key missing")
    need("Possible response" not in checks and "Possible" in key, "Learner copy leaks worked solution")
    need("PLAN ONLY" in checks and "not secure" in checks and "term grade" in key, "Assessment boundary missing")
    need(swaps.count("AFTER first check capture only") == 4, "Check-day worked models must follow first capture")
    need("## Parent/carer pass" in read("RUN-THROUGH.md"), "Parent desk pass missing")
    need("Science" in lessons and "real plant" in checks and "family" in checks, "Cross-subject evidence boundary missing")
    print("PASS ten timed 25-minute scripts, 30 routes, 20 worked swaps and two held checks")


def assets() -> None:
    alternatives = read("print/TEXT-ALTERNATIVES.md")
    need(len(STEMS) == 8 and len(list((ROOT / "print").glob("*.svg"))) == 8 and len(list((ROOT / "print").glob("*.pdf"))) == 8, "Eight pairs missing or extra")
    for stem in STEMS:
        svg = ROOT / "print" / f"{stem}.svg"
        pdf = ROOT / "print" / f"{stem}.pdf"
        xml = ET.parse(svg).getroot()
        need(xml.attrib.get("width") == "210mm" and xml.attrib.get("height") == "297mm" and xml.attrib.get("viewBox") == "0 0 794 1123", f"A4 SVG geometry mismatch: {stem}")
        need(xml.find("{http://www.w3.org/2000/svg}title") is not None and xml.find("{http://www.w3.org/2000/svg}desc") is not None, f"SVG text alternative missing: {stem}")
        pages = PdfReader(pdf).pages
        need(len(pages) == 1 and abs(float(pages[0].mediabox.width) - 595.28) < 1 and abs(float(pages[0].mediabox.height) - 841.89) < 1, f"A4 PDF mismatch: {stem}")
        extracted = pages[0].extract_text()
        compact_text = re.sub(r"\s+", "", extracted)
        need(all(re.sub(r"\s+", "", phrase) in compact_text for phrase in PDF_PHRASES[stem]), f"PDF text mismatch: {stem}")
        bbox = subprocess.run(["pdftotext", "-bbox", str(pdf), "-"], check=True, capture_output=True, text=True).stdout
        bbox_root = ET.fromstring(bbox)
        for item in bbox_root.iter():
            if item.tag.endswith("}word"):
                need(0 <= float(item.attrib["xMin"]) < float(item.attrib["xMax"]) <= 595.3 and 0 <= float(item.attrib["yMin"]) < float(item.attrib["yMax"]) <= 841.9, f"Clipped PDF text: {stem}")
        need(f"({stem}.pdf)" in alternatives and f"({stem}.svg)" in alternatives, f"Alternative pair link missing: {stem}")
    need("not tagged" in alternatives.lower() and "tactile" in alternatives.lower() and "no print/touch" in alternatives.lower(), "Linear/tactile/no-print boundary absent")
    namespace = {"s": "http://www.w3.org/2000/svg"}
    lantern = ET.parse(ROOT / "print/check-u-lantern.svg").getroot()
    opening = next(node for node in lantern.findall("s:rect", namespace) if node.get("data-part") == "wide rectangular opening")
    need(float(opening.get("width")) > float(opening.get("height")), "Check U wide opening contradicts source")
    need(len(lantern.findall("s:path", namespace)) == 1, "Check U must have no worked surface lines")
    edge = ET.parse(ROOT / "print/check-v-edge.svg").getroot()
    paths = [node.get("d") for node in edge.findall("s:path", namespace) if "Q" in node.get("d", "")]
    need(len(paths) == 2 and all(value.count("Q") == 4 for value in paths), "Check V three-bump edge geometry drift")
    coordinates = [[float(value) for value in re.findall(r"-?\d+(?:\.\d+)?", path)] for path in paths]
    need(all(right - left == (362 if index % 2 == 0 else 0) for index, (left, right) in enumerate(zip(*coordinates))), "Check V edge cards must be identical translations")
    print("PASS eight A4 SVG/PDF pairs, searchable text and exact access alternatives")


def regeneration() -> None:
    with tempfile.TemporaryDirectory(prefix="subjectnest-va-w13-14-") as tmp:
        temp = Path(tmp)
        shutil.copy2(ROOT / "print/generate_print.py", temp / "generate_print.py")
        subprocess.run([sys.executable, str(temp / "generate_print.py")], check=True, capture_output=True)
        for stem in STEMS:
            for ext in ("svg", "pdf"):
                name = f"{stem}.{ext}"
                need(hashlib.sha256((temp / name).read_bytes()).digest() == hashlib.sha256((ROOT / "print" / name).read_bytes()).digest(), f"Nondeterministic or stale asset: {name}")
    print("PASS deterministic regeneration of all sixteen A4 assets")


def digital() -> None:
    html = read("digital/index.html")
    js = read("digital/app.js")
    css = read("digital/style.css")
    guide = read("digital/README.md")
    need("Content-Security-Policy" in html and "default-src 'none'" in html and '<script src="app.js" defer>' in html and '<link rel="stylesheet" href="style.css">' in html, "Offline browser scaffold drift")
    need(all(f'id="{name}"' in html for name in ("length", "direction", "spacing", "board", "description", "save-a", "save-b", "clear", "idea-a", "idea-b", "status")), "Browser controls missing")
    need("aria-live=\"polite\"" in html and ":focus-visible" in css and "addEventListener" in js and "textContent" in js, "Browser access or safe text route missing")
    need(not re.search(r"\b(?:fetch|XMLHttpRequest|localStorage|sessionStorage|sendBeacon|WebSocket|serviceWorker)\b", js), "Unexpected browser network/storage API")
    need("closing or reloading clears" in (html + js).lower() and "screen" in read("MATERIALS.md").lower(), "Privacy or no-screen boundary missing")
    subprocess.run(["node", "--check", str(ROOT / "digital/app.js")], check=True, capture_output=True)
    for name, width in (("desktop", 1280), ("mobile", 390)):
        preview = ROOT / "digital/previews" / f"{name}.png"
        raw = preview.read_bytes()
        need(raw[:8] == b"\x89PNG\r\n\x1a\n" and struct.unpack(">I", raw[16:20])[0] == width and f"previews/{name}.png" in guide, f"Browser preview missing: {name}")
    need("1280, 390 and 320" in guide and (ROOT / "digital/qa_browser.cjs").is_file(), "Substantive browser QA route missing")
    print("PASS offline text/keyboard line-texture lab, JS syntax and two preview dimensions")


def slug(title: str) -> str:
    title = re.sub(r"\[([^]]+)\]\([^)]+\)", r"\1", title)
    title = re.sub(r"[*_]", "", title).strip().lower()
    return re.sub(r"[^\w -]", "", title).replace(" ", "-")


def links(*, bootstrap: bool) -> None:
    count = 0
    for path in ROOT.rglob("*.md"):
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("https://", "http://", "mailto:")):
                continue
            name, marker, anchor = unquote(url).partition("#")
            target = (path.parent / name).resolve() if name else path
            if bootstrap and target == ROOT / "manifest.json":
                count += 1
                continue
            need(target.is_file(), f"Broken local link: {path.relative_to(ROOT)} -> {url}")
            if marker and target.suffix == ".md":
                headings = [slug(value) for value in re.findall(r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)]
                need(anchor in headings, f"Broken heading anchor: {url}")
            count += 1
    print(f"PASS {count} local links and heading anchors")


def manifest_data() -> dict:
    files = sorted(path for path in ROOT.rglob("*") if path.is_file() and path.name != "manifest.json" and "__pycache__" not in path.parts)
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/visual-arts/term-2/weeks-13-14",
        "created_at": "2026-09-30",
        "pack_version": "0.1.0-draft",
        "review_status": "author_desk_checked_pending_school_child_accessibility_and_cultural_review",
        "curriculum_source_sha256": SOURCE_SHA,
        "curriculum_codes_partial_conditional": sorted(ROWS),
        "contextual_codes_separate_evidence": sorted(CONTEXT_ROWS),
        "files": {str(path.relative_to(ROOT)): {"sha256": hashlib.sha256(path.read_bytes()).hexdigest(), "bytes": path.stat().st_size} for path in files},
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true", help="One-time authoring update; default is read-only")
    args = parser.parse_args()
    source()
    pedagogy()
    assets()
    regeneration()
    digital()
    links(bootstrap=args.write_manifest)
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
        print(f"WROTE SHA-256 manifest for {len(expected['files'])} files")
    else:
        need(json.loads(manifest.read_text(encoding="utf-8")) == expected, "SHA-256 manifest missing or stale")
        print(f"PASS SHA-256 manifest for {len(expected['files'])} files")
    print("PACK PASS · author desk QA only; no child or classroom trial")


if __name__ == "__main__":
    main()
