# SPDX-License-Identifier: Apache-2.0
"""Read-only curriculum, lesson, asset and receipt checks for Visual Arts W5-6."""

from __future__ import annotations

import argparse
import ast
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
ROWS = {"AC9AVAFE01": 21465, "AC9AVAFD01": 21474, "AC9AVAFC01": 21484, "AC9AVAFP01": 21493}
DESCRIPTIONS = {
    "AC9AVAFE01": "explore how and why the arts are important for people and communities",
    "AC9AVAFD01": "use play, imagination, arts knowledge, processes and/or skills to discover possibilities and develop ideas",
    "AC9AVAFC01": "create arts works that communicate ideas",
    "AC9AVAFP01": "share their arts works with audiences",
}
QCAA_URL = "https://www.qcaa.qld.edu.au/p-10/aciq/version-9/learning-areas/p-10-the-arts/visual-arts"
QCAA_ALIGNMENT_URL = "https://www.qcaa.qld.edu.au/downloads/aciqv9/the-arts/curriculum/ac9_visual_arts_prep_as_cd.alignment.pdf"
QCAA_TECHNIQUES_URL = "https://www.qcaa.qld.edu.au/downloads/aciqv9/the-arts/assessment/ac9_arts_non_perform_tc_prep.pdf"
STEMS = (
    "source-choice", "contour-silhouette", "gap-between", "layer-order",
    "material-choice", "artwork-record", "check-k-paper-fan", "check-l-robot-postcard",
)
EXPECTED_CODES = {
    21: {"AC9AVAFD01"},
    22: {"AC9AVAFD01", "AC9AVAFC01"},
    23: {"AC9AVAFD01", "AC9AVAFC01"},
    24: {"AC9AVAFD01", "AC9AVAFC01"},
    25: {"AC9AVAFD01", "AC9AVAFC01"},
    26: {"AC9AVAFD01", "AC9AVAFC01"},
    27: {"AC9AVAFD01", "AC9AVAFC01"},
    28: {"AC9AVAFD01", "AC9AVAFC01"},
    29: {"AC9AVAFE01", "AC9AVAFD01", "AC9AVAFP01"},
    30: {"AC9AVAFD01", "AC9AVAFC01"},
}
PDF_PHRASES = {
    "source-choice": ("INVENTED FORK FORM", "NOT A REAL PLANT", "REAL SOURCE: EMPTY", "three rounded arms", "two open notches"),
    "contour-silhouette": ("CONTOUR", "FILLED SILHOUETTE", "Outside line", "Body filled"),
    "gap-between": ("NARROW GAP", "WIDE GAP", "same", "opening between edges"),
    "layer-order": ("CURVED IN FRONT", "ANGLED IN FRONT", "edge is hidden", "Drawn overlap"),
    "material-choice": ("BROAD PAPER", "BROAD CARD", "ACTUAL PROPERTIES: UNKNOWN", "adult-approved real inspection"),
    "artwork-record": ("MY IDEA / INTENDED VIEWER", "TWO VISUAL OPTIONS / CHOICE", "ACTUAL MAKER / MATERIAL ROUTE", "WILLING VIEWER / NEXT QUESTION"),
    "check-k-paper-fan": ("INVENTED PAPER FAN FORM", "four rays", "middle open wedge", "IDEA 1", "IDEA 2"),
    "check-l-robot-postcard": ("TWO SEPARATE PIECES", "ROUND HEAD", "STEP-SHAPED TOOL", "IDEA 1", "IDEA 2"),
}


def need(ok: object, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def read(name: str) -> str:
    return (ROOT / name).read_text(encoding="utf-8")


def source() -> None:
    workbook = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    imported = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8"))
    snapshot = json.loads(read("source-snapshot.json"))
    crosswalk = read("CURRICULUM-CROSSWALK.md")
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == imported["source_sha256"]
        == snapshot["workbook_sha256"]
        == SOURCE_SHA,
        "Official ACARA workbook/import/snapshot SHA drift",
    )
    need(snapshot["content_rows"] == ROWS, "Official row pin drift")
    need(
        snapshot["queensland_prep_entry_point_url"] == QCAA_URL
        and snapshot["queensland_prep_alignment_url"] == QCAA_ALIGNMENT_URL
        and snapshot["queensland_prep_techniques_url"] == QCAA_TECHNIQUES_URL
        and all(url in crosswalk for url in (QCAA_URL, QCAA_ALIGNMENT_URL, QCAA_TECHNIQUES_URL)),
        "Queensland Prep primary source link drift",
    )
    records = {
        item["code"]: item for item in imported["records"]
        if item.get("record_type") == "content_description" and item.get("code") in ROWS
    }
    need(set(records) == set(ROWS), "Foundation Visual Arts descriptions missing")
    for code, row in ROWS.items():
        item = records[code]
        attrs = item["attributes"]
        need(
            item["source_row"] == row
            and item["plain_text"] == DESCRIPTIONS[code]
            and attrs["learning_area"] == "The Arts"
            and attrs["subject"] == "Visual Arts"
            and attrs["level"] == "Foundation Year",
            f"Exact official import drift: {code}",
        )
        pattern = rf"^\| {code} \| {row} \| The Arts · Visual Arts · Foundation Year \| {re.escape(DESCRIPTIONS[code])} \|"
        need(re.search(pattern, crosswalk, re.MULTILINE), f"Exact crosswalk drift: {code}")
    need("school-selected" in crosswalk.lower() and "not secure exams" in crosswalk.lower() and "AC9ADVAFE01" in crosswalk, "Source/claim/erratum boundary missing")
    print("PASS four exact Foundation Visual Arts workbook rows and Queensland Prep primary links")


def pedagogy() -> None:
    lessons = read("LESSONS.md")
    headings = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", lessons, re.MULTILINE))
    need([int(m.group(2)) for m in headings] == list(range(21, 31)), "Teacher day sequence drift")
    for index, mark in enumerate(headings):
        day = int(mark.group(2))
        end = headings[index + 1].start() if index + 1 < len(headings) else len(lessons)
        body = lessons[mark.end():end]
        minutes = [int(v) for v in re.findall(r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*", body, re.MULTILINE)]
        need(int(mark.group(1)) == (5 if day <= 25 else 6) and minutes == [3, 4, 5, 7, 4, 2] and sum(minutes) == 25, f"Day {day} week/timing drift: {minutes}")
        target = re.search(r"^\*\*Target/codes:\*\* ([^\n]+)", body, re.MULTILINE)
        need(target is not None, f"Day {day} target missing")
        codes = set(re.findall(r"\bAC9AVA(?:FE|FD|FC|FP)01\b", target.group(1)))
        need(codes == EXPECTED_CODES[day], f"Day {day} official code drift: {codes}")
        need("**Prepare:**" in body and ("**Notice · 4 min.**" in body or "**Self-check · 4 min.**" in body), f"Day {day} material/formative step missing")
    cards = read("LEARNER-CARDS.md")
    card_headings = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need([int(m.group(1)) for m in card_headings] == list(range(21, 31)), "Learner day sequence drift")
    for index, mark in enumerate(card_headings):
        end = card_headings[index + 1].start() if index + 1 < len(card_headings) else len(cards)
        body = cards[mark.end():end]
        need("**Shared target:**" in body and all(re.search(rf"^- \*\*{route} ·", body, re.MULTILINE) for route in "ABC") and "**Home:**" in body, f"Day {mark.group(1)} route/home drift")
    swaps = re.findall(r"^\| (\d+) ·[^|]*\| ([^|]+) \| ([^|]+) \|$", read("PRACTICE-SWAPS.md"), re.MULTILINE)
    need([int(day) for day, _, _ in swaps] == list(range(21, 31)) and all(a.startswith("**") and b.startswith("**") for _, a, b in swaps), "Twenty worked domain swaps missing")
    need("local approval" in cards.lower() and "no child cutting" in read("MATERIALS.md").lower() and "unknown" in lessons.lower(), "Access/safety/source boundary missing")
    print("PASS ten 25-minute scripts, 30 switchable routes and 20 worked bridges")


def checks() -> None:
    child = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    routine = read("LEARNER-CARDS.md") + read("PRACTICE-SWAPS.md")
    need(re.findall(r"^## Check ([KL]) · Day (\d+)\b", child, re.MULTILINE) == [("K", "25"), ("L", "30")], "Check schedule drift")
    need("teacher/KEY-AND-NEXT.md" in child and "not secure tests" in child.lower() and "open after first response" in key.lower(), "Public check/key boundary missing")
    for held in ("four broad rays", "two middle rays", "ROUND HEAD", "STEP-SHAPED TOOL"):
        need(held not in routine, f"Held assessment detail leaked into routine: {held}")
    for phrase in ("low broad curved base", "four broad rays", "open wedge", "ROUND HEAD", "STEP-SHAPED TOOL", "UNKNOWN/NOT TRIED", "actual", "separate"):
        need(phrase.lower() in key.lower(), f"Teacher key missing: {phrase}")
    need("fictional" in child.lower() or "invented" in child.lower(), "Fresh invented source boundary absent")
    print("PASS two fresh public checks, independent first response and separate educator key")


def data_values(stem: str, attribute: str) -> list[str]:
    top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
    ns = {"svg": "http://www.w3.org/2000/svg"}
    return [node.get(attribute, "") for node in top.findall(".//svg:rect", ns) if node.get(attribute) is not None]


def assets() -> None:
    alternatives = read("print/TEXT-ALTERNATIVES.md")
    source_tree = ast.parse(read("print/generate_print.py"))
    pages = next(node.value for node in source_tree.body if isinstance(node, ast.Assign) and any(isinstance(t, ast.Name) and t.id == "PAGES" for t in node.targets))
    need(isinstance(pages, ast.Dict) and {k.value for k in pages.keys if isinstance(k, ast.Constant)} == set(STEMS), "Generator inventory drift")
    ns = {"svg": "http://www.w3.org/2000/svg"}
    alt_flat = " ".join(alternatives.lower().split())
    for stem in STEMS:
        top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
        pdf = ROOT / "print" / f"{stem}.pdf"
        need(top.get("width") == "210mm" and top.get("height") == "297mm" and top.find("svg:title", ns) is not None and top.find("svg:desc", ns) is not None, f"SVG metadata/A4 drift: {stem}")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        words = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        pdf_flat = " ".join(words.lower().split())
        need(re.search(r"Pages:\s+1\b", info) and "(A4)" in info and "CC BY 4.0" in words and len(words.split()) >= 25 and f"{stem}.svg" in alternatives and f"{stem}.pdf" in alternatives, f"PDF/text/alternative drift: {stem}")
        for phrase in PDF_PHRASES[stem]:
            need(phrase.lower() in pdf_flat and phrase.lower() in alt_flat, f"PDF/linear mismatch: {stem}: {phrase}")
    need(
        data_values("source-choice", "data-source") == ["INVENTED FORK FORM", "REAL SOURCE EMPTY"]
        and data_values("contour-silhouette", "data-treatment") == ["CONTOUR", "FILLED SILHOUETTE"]
        and data_values("gap-between", "data-gap") == ["NARROW GAP", "WIDE GAP"]
        and data_values("layer-order", "data-front") == ["CURVED PIECE", "ANGLED PIECE"]
        and data_values("material-choice", "data-candidate") == ["BROAD PAPER", "BROAD CARD"]
        and len(data_values("artwork-record", "data-field")) == 4
        and data_values("check-k-paper-fan", "data-canvas") == ["IDEA 1", "IDEA 2"]
        and data_values("check-l-robot-postcard", "data-canvas") == ["IDEA 1", "IDEA 2"],
        "Original source/order metadata drift",
    )
    need("not tagged" in alternatives.lower() and "tactile" in alternatives.lower() and "no material" in alternatives.lower(), "Accessible equivalent route absent")
    print("PASS eight original single-page A4 SVG/PDF pairs with exact linear/tactile/no-material routes")


def slug(title: str) -> str:
    title = re.sub(r"\[([^]]+)\]\([^)]+\)", r"\1", title)
    title = re.sub(r"[*_]", "", title).strip().lower()
    return re.sub(r"[^\w -]", "", title).replace(" ", "-")


def links(*, bootstrap: bool) -> None:
    count = 0
    for path in ROOT.rglob("*.md"):
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("https://", "http://", "mailto:")):
                continue
            filename, marker, anchor = unquote(url).partition("#")
            target = (path.parent / filename).resolve() if filename else path
            if bootstrap and target == ROOT / "manifest.json":
                count += 1
                continue
            need(target.is_file(), f"Broken local link: {path.relative_to(ROOT)} -> {url}")
            if marker and target.suffix == ".md":
                headings = [slug(text) for text in re.findall(r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)]
                need(anchor in headings, f"Broken heading anchor: {url}")
            count += 1
    print(f"PASS {count} local file/heading links")


def manifest_data() -> dict:
    files = sorted(path for path in ROOT.rglob("*") if path.is_file() and path.name != "manifest.json" and "__pycache__" not in path.parts)
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/visual-arts/term-1/weeks-05-06",
        "created_at": "2026-09-30",
        "review_status": "author_desk_checked_pending_school_material_child_accessibility_and_cultural_review",
        "curriculum_source_sha256": SOURCE_SHA,
        "curriculum_codes_partial_conditional": sorted(ROWS),
        "files": {str(path.relative_to(ROOT)): {"sha256": hashlib.sha256(path.read_bytes()).hexdigest(), "bytes": path.stat().st_size} for path in files},
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    source()
    pedagogy()
    checks()
    assets()
    links(bootstrap=args.write_manifest)
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
        print(f"WROTE SHA-256 manifest for {len(expected['files'])} files")
    else:
        need(json.loads(manifest.read_text(encoding="utf-8")) == expected, "SHA-256 manifest missing or stale")
        print(f"PASS SHA-256 manifest for {len(expected['files'])} files")
    print("PACK PASS · author desk QA only; no child or physical trial")


if __name__ == "__main__":
    main()
