# SPDX-License-Identifier: Apache-2.0
"""Read-only source, lesson, media, link and SHA checks for Visual Arts W7-8."""

from __future__ import annotations

import argparse
import ast
import hashlib
import json
import re
import struct
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
ROWS = {"AC9AVAFE01": 21465, "AC9AVAFD01": 21474, "AC9AVAFC01": 21484, "AC9AVAFP01": 21493}
DESCRIPTIONS = {
    "AC9AVAFE01": "explore how and why the arts are important for people and communities",
    "AC9AVAFD01": "use play, imagination, arts knowledge, processes and/or skills to discover possibilities and develop ideas",
    "AC9AVAFC01": "create arts works that communicate ideas",
    "AC9AVAFP01": "share their arts works with audiences",
}
MATH_CODE, MATH_ROW = "AC9MFA01", 17181
MATH_DESCRIPTION = "recognise, copy and continue repeating patterns represented in different ways"
QCAA_URL = "https://www.qcaa.qld.edu.au/p-10/aciq/version-9/learning-areas/p-10-the-arts/visual-arts"
QCAA_ALIGNMENT = "https://www.qcaa.qld.edu.au/downloads/aciqv9/the-arts/curriculum/ac9_visual_arts_prep_as_cd.alignment.pdf"
QCAA_TECHNIQUES = "https://www.qcaa.qld.edu.au/downloads/aciqv9/the-arts/assessment/ac9_arts_non_perform_tc_prep.pdf"
STEMS = (
    "unknown-cue", "two-panel-path", "sequence-repeat", "visual-unit",
    "quiet-transfer", "art-viewer-record", "check-m-shelf-card", "check-n-path-strip",
)
EXPECTED_CODES = {
    31: {"AC9AVAFD01"},
    32: {"AC9AVAFD01", "AC9AVAFC01"},
    33: {"AC9AVAFD01", "AC9AVAFC01"},
    34: {"AC9AVAFD01", "AC9AVAFC01"},
    35: {"AC9AVAFD01", "AC9AVAFC01"},
    36: {"AC9AVAFD01"},
    37: {"AC9AVAFD01", "AC9AVAFC01"},
    38: {"AC9AVAFD01", "AC9AVAFC01"},
    39: {"AC9AVAFE01", "AC9AVAFD01", "AC9AVAFC01", "AC9AVAFP01"},
    40: {"AC9AVAFD01", "AC9AVAFC01"},
}
PDF_PHRASES = {
    "unknown-cue": ("FICTIONAL PAPER PLANET SHELF", "LOOP above SQUARE", "MEANING UNKNOWN", "no real sign"),
    "two-panel-path": ("SAME TWO BLANK PANELS", "ACROSS", "DOWN", "NOTICE", "ASK"),
    "sequence-repeat": ("STORY CHANGE", "OPEN BOX", "CLOSED BOX", "REPEAT: ARCH, BAR | ARCH, BAR"),
    "visual-unit": ("ONE UNIT: ARCH THEN BAR", "UNIT 1", "UNIT 2", "UNIT 3", "BETWEEN"),
    "quiet-transfer": ("1 VISUAL MARKS", "2 QUIET CUES", "3 RETURN TO MARKS", "HAND UP then STILL"),
    "art-viewer-record": ("MY IMAGE IDEA / FIRST VIEW", "TWO OPTIONS / ONE CHANGE", "ACTUAL MAKER / SUPPORT", "ACTUAL WILLING VIEWER"),
    "check-m-shelf-card": ("TEARDROP left of OPEN BRACKET", "MEANING UNKNOWN", "IDEA 1", "IDEA 2"),
    "check-n-path-strip": ("BENT LINE", "SMALL BLOCK", "MY TWO-MARK UNIT", "IDEA 1", "IDEA 2"),
}


def need(ok: object, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def read(name: str) -> str:
    return (ROOT / name).read_text(encoding="utf-8")


def source() -> None:
    workbook = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    imported = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8"))
    snapshot = json.loads(read("source-snapshot.json"))
    crosswalk = read("CURRICULUM-CROSSWALK.md")
    need(hashlib.sha256(workbook.read_bytes()).hexdigest() == imported["source_sha256"] == snapshot["workbook_sha256"] == SOURCE_SHA, "Official workbook/import/snapshot SHA drift")
    need(snapshot["content_rows"] == ROWS and snapshot["contextual_mathematics_row"] == {MATH_CODE: MATH_ROW}, "Official source row pin drift")
    need(snapshot["queensland_prep_entry_point_url"] == QCAA_URL and snapshot["queensland_prep_alignment_url"] == QCAA_ALIGNMENT and snapshot["queensland_prep_techniques_url"] == QCAA_TECHNIQUES and all(url in crosswalk for url in (QCAA_URL, QCAA_ALIGNMENT, QCAA_TECHNIQUES)), "Queensland Prep primary link drift")
    codes = set(ROWS) | {MATH_CODE}
    records = {item["code"]: item for item in imported["records"] if item.get("record_type") == "content_description" and item.get("code") in codes}
    need(set(records) == codes, "Official Visual Arts/contextual maths descriptions missing")
    for code, row in ROWS.items():
        item = records[code]
        attrs = item["attributes"]
        need(item["source_row"] == row and item["plain_text"] == DESCRIPTIONS[code] and attrs["learning_area"] == "The Arts" and attrs["subject"] == "Visual Arts" and attrs["level"] == "Foundation Year", f"Official Visual Arts import drift: {code}")
        need(re.search(rf"^\| {code} \| {row} \| The Arts · Visual Arts · Foundation Year \| {re.escape(DESCRIPTIONS[code])} \|", crosswalk, re.MULTILINE), f"Exact crosswalk drift: {code}")
    math = records[MATH_CODE]
    need(math["source_row"] == MATH_ROW and math["plain_text"] == MATH_DESCRIPTION and math["attributes"]["learning_area"] == "Mathematics" and math["attributes"]["level"] == "Foundation Year" and MATH_CODE in crosswalk and MATH_DESCRIPTION in crosswalk, "Contextual maths source drift")
    need("school-selected" in crosswalk.lower() and "not secure exams" in crosswalk.lower() and "AC9ADVAFE01" in crosswalk and "not an official safety sign" in crosswalk.lower(), "Source/claim/erratum boundary missing")
    print("PASS exact four Visual Arts rows, contextual maths row and Queensland Prep source pins")


def pedagogy() -> None:
    lessons = read("LESSONS.md")
    headings = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", lessons, re.MULTILINE))
    need([int(m.group(2)) for m in headings] == list(range(31, 41)), "Teacher day sequence drift")
    for index, mark in enumerate(headings):
        day = int(mark.group(2))
        end = headings[index + 1].start() if index + 1 < len(headings) else len(lessons)
        body = lessons[mark.end():end]
        minutes = [int(v) for v in re.findall(r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*", body, re.MULTILINE)]
        need(int(mark.group(1)) == (7 if day <= 35 else 8) and minutes == [3, 4, 5, 7, 4, 2] and sum(minutes) == 25, f"Day {day} timing/week drift: {minutes}")
        target = re.search(r"^\*\*Target/codes:\*\* ([^\n]+)", body, re.MULTILINE)
        need(target is not None and "**Prepare:**" in body, f"Day {day} target/prepare missing")
        codes = set(re.findall(r"\bAC9AVA(?:FE|FD|FC|FP)01\b", target.group(1)))
        need(codes == EXPECTED_CODES[day], f"Day {day} curriculum code drift: {codes}")
        need("**Notice · 4 min.**" in body or "**Self-check · 4 min.**" in body, f"Day {day} formative step missing")
    cards = read("LEARNER-CARDS.md")
    card_headings = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need([int(m.group(1)) for m in card_headings] == list(range(31, 41)), "Learner day sequence drift")
    for index, mark in enumerate(card_headings):
        end = card_headings[index + 1].start() if index + 1 < len(card_headings) else len(cards)
        body = cards[mark.end():end]
        need("**Shared target:**" in body and all(re.search(rf"^- \*\*{route} ·", body, re.MULTILINE) for route in "ABC") and "**Family bridge:**" in body, f"Day {mark.group(1)} route/family drift")
    swaps = re.findall(r"^\| (\d+) ·[^|]*\| ([^|]+) \| ([^|]+) \|$", read("PRACTICE-SWAPS.md"), re.MULTILINE)
    need([int(day) for day, _, _ in swaps] == list(range(31, 41)) and all(a.startswith("**") and b.startswith("**") for _, a, b in swaps), "Twenty worked swaps missing")
    need("not fixed" in cards.lower() and "plan only" in lessons.lower() and "no child cutting" in read("MATERIALS.md").lower(), "Access/safety boundaries missing")
    print("PASS ten 25-minute scripts, 30 routes, ten family bridges and 20 worked swaps")


def checks() -> None:
    child = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    routine = read("LEARNER-CARDS.md") + read("PRACTICE-SWAPS.md") + read("digital/README.md") + read("digital/index.html")
    need(re.findall(r"^## Check ([MN]) · Day (\d+)\b", child, re.MULTILINE) == [("M", "35"), ("N", "40")], "Check schedule drift")
    need("teacher/KEY-AND-NEXT.md" in child and "not secure tests" in child.lower() and "open after first response" in key.lower(), "Public check/key boundary missing")
    for held in ("TEARDROP", "BRACKET", "BENT LINE", "SMALL BLOCK"):
        need(held not in routine, f"Held source leaked into daily/digital routine: {held}")
    for phrase in ("meaning **UNKNOWN**", "two different", "BENT LINE", "SMALL BLOCK", "PLAN ONLY", "actual", "mathematics"):
        need(phrase.lower() in key.lower(), f"Educator key missing: {phrase}")
    print("PASS two fresh learner-clean checks and separate next-teaching key")


def data_values(stem: str, attribute: str) -> list[str]:
    top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
    ns = {"svg": "http://www.w3.org/2000/svg"}
    return [node.get(attribute, "") for node in top.findall(".//svg:rect", ns) if node.get(attribute) is not None]


def assets() -> None:
    alternatives = read("print/TEXT-ALTERNATIVES.md")
    source_tree = ast.parse(read("print/generate_print.py"))
    pages = next(node.value for node in source_tree.body if isinstance(node, ast.Assign) and any(isinstance(t, ast.Name) and t.id == "PAGES" for t in node.targets))
    need(isinstance(pages, ast.Dict) and {k.value for k in pages.keys if isinstance(k, ast.Constant)} == set(STEMS), "Generator inventory drift")
    ns = {"svg": "http://www.w3.org/2000/svg"}
    alt_flat = " ".join(alternatives.lower().split())
    for stem in STEMS:
        top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
        pdf = ROOT / "print" / f"{stem}.pdf"
        need(top.get("width") == "210mm" and top.get("height") == "297mm" and top.find("svg:title", ns) is not None and top.find("svg:desc", ns) is not None, f"SVG metadata/A4 drift: {stem}")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        words = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        pdf_flat = " ".join(words.lower().split())
        need(re.search(r"Pages:\s+1\b", info) and "(A4)" in info and "CC BY 4.0" in words and len(words.split()) >= 28 and f"{stem}.svg" in alternatives and f"{stem}.pdf" in alternatives, f"PDF/text/alternative drift: {stem}")
        for phrase in PDF_PHRASES[stem]:
            need(phrase.lower() in pdf_flat and phrase.lower() in alt_flat, f"PDF/linear mismatch: {stem}: {phrase}")
    need(data_values("unknown-cue", "data-source") == ["LOOP-SQUARE UNKNOWN"] and data_values("two-panel-path", "data-layout") == ["ACROSS", "DOWN"] and data_values("sequence-repeat", "data-structure") == ["STORY CHANGE", "REPEAT"] and data_values("visual-unit", "data-unit") == ["1", "2", "3"] and data_values("quiet-transfer", "data-stage") == ["1 VISUAL MARKS", "2 QUIET CUES", "3 RETURN TO MARKS"] and len(data_values("art-viewer-record", "data-field")) == 4 and data_values("check-m-shelf-card", "data-canvas") == ["IDEA 1", "IDEA 2"] and data_values("check-n-path-strip", "data-canvas") == ["MY UNIT", "IDEA 1", "IDEA 2"], "Source/order metadata drift")
    need("not tagged" in alternatives.lower() and "tactile" in alternatives.lower() and "no material" in alternatives.lower(), "Equivalent access routes missing")
    print("PASS eight original A4 SVG/PDF pairs and exact linear/tactile/no-material routes")


def digital() -> None:
    html = read("digital/index.html")
    js = read("digital/app.js")
    css = read("digital/style.css")
    guide = read("digital/README.md")
    need("Content-Security-Policy" in html and "default-src 'none'" in html and "aria-live=\"polite\"" in html and "<script src=\"app.js\" defer>" in html and "<link rel=\"stylesheet\" href=\"style.css\">" in html, "Offline/access browser scaffold drift")
    need(all(f'id="{name}"' in html for name in ("layout", "order", "cue", "panels", "mark-a", "mark-b", "unit", "space", "beat", "status")), "Browser control drift")
    need("replaceChildren" in js and "addEventListener" in js and "aria-label" in js and ":focus-visible" in css, "Browser interaction/access code drift")
    need(not re.search(r"\b(?:fetch|XMLHttpRequest|localStorage|sessionStorage|sendBeacon|WebSocket|serviceWorker)\b", js) and "No choice is saved or sent" in html + js and "No choice is saved or sent" in guide + html + js, "Unexpected storage/network/claim drift")
    for name, width in (("desktop", 1280), ("mobile", 390)):
        preview = ROOT / "digital" / "previews" / f"{name}.png"
        pixels = preview.read_bytes()
        need(pixels[:8] == b"\x89PNG\r\n\x1a\n" and struct.unpack(">I", pixels[16:20])[0] == width and f"previews/{name}.png" in guide, f"Review preview missing or wrong size: {name}")
    need("offline" in guide.lower() and "screen" in guide.lower(), "Optional route boundary missing")
    print("PASS optional local-only browser scaffold, text controls and privacy boundary")


def slug(title: str) -> str:
    title = re.sub(r"\[([^]]+)\]\([^)]+\)", r"\1", title)
    title = re.sub(r"[*_]", "", title).strip().lower()
    return re.sub(r"[^\w -]", "", title).replace(" ", "-")


def links(*, bootstrap: bool) -> None:
    count = 0
    for path in ROOT.rglob("*.md"):
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("https://", "http://", "mailto:")):
                continue
            filename, marker, anchor = unquote(url).partition("#")
            target = (path.parent / filename).resolve() if filename else path
            if bootstrap and target == ROOT / "manifest.json":
                count += 1
                continue
            need(target.is_file(), f"Broken local link: {path.relative_to(ROOT)} -> {url}")
            if marker and target.suffix == ".md":
                headings = [slug(text) for text in re.findall(r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE)]
                need(anchor in headings, f"Broken heading anchor: {url}")
            count += 1
    print(f"PASS {count} local links and heading anchors")


def manifest_data() -> dict:
    files = sorted(path for path in ROOT.rglob("*") if path.is_file() and path.name != "manifest.json" and "__pycache__" not in path.parts)
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/visual-arts/term-1/weeks-07-08",
        "created_at": "2026-09-30",
        "review_status": "author_desk_checked_pending_school_child_accessibility_and_cultural_review",
        "curriculum_source_sha256": SOURCE_SHA,
        "curriculum_codes_partial_conditional": sorted(ROWS),
        "contextual_mathematics_code": MATH_CODE,
        "files": {str(path.relative_to(ROOT)): {"sha256": hashlib.sha256(path.read_bytes()).hexdigest(), "bytes": path.stat().st_size} for path in files},
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    source()
    pedagogy()
    checks()
    assets()
    digital()
    links(bootstrap=args.write_manifest)
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
        print(f"WROTE SHA-256 manifest for {len(expected['files'])} files")
    else:
        need(json.loads(manifest.read_text(encoding="utf-8")) == expected, "SHA-256 manifest missing or stale")
        print(f"PASS SHA-256 manifest for {len(expected['files'])} files")
    print("PACK PASS · author desk QA only; no child or classroom trial")


if __name__ == "__main__":
    main()
