#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only source, lesson, print, interactive and hash verification."""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import tempfile
import warnings
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

import openpyxl
from PIL import ImageFont

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = {"AC9ADAFE01", "AC9ADAFD01", "AC9ADAFC01", "AC9ADAFP01"}
STEMS = (
    "two-move-key",
    "repeat-unit",
    "deliberate-ending",
    "maker-plan",
    "response-note",
    "supplied-work",
    "fresh-m",
    "fresh-n",
)

TEACHING = (
    "README.md",
    "LESSONS.md",
    "MATERIALS.md",
    "LEARNER-CARDS.md",
    "OPTIONAL-PRACTICE.md",
    "READ-ALOUD.md",
    "FAMILY-OPTIONAL.md",
)


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def sources() -> None:
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    data = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text())
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == SOURCE_SHA
        == data["source_sha256"],
        "Official source hash mismatch",
    )
    rows = {
        r["code"]: r
        for r in data["records"]
        if r.get("record_type") == "content_description" and r.get("code") in CODES
    }
    table = {}
    for line in (ROOT / "CURRICULUM-CROSSWALK.md").read_text().splitlines():
        m = re.match(
            r"^\| (AC9[A-Z0-9]+) \| (\d+) \| Dance · Foundation Year \| (.*?) \|$",
            line,
        )
        if m:
            table[m[1]] = (int(m[2]), m[3])
    need(set(rows) == set(table) == CODES, "Wrong crosswalk codes")
    with warnings.catch_warnings():
        warnings.simplefilter("ignore", UserWarning)
        source = openpyxl.load_workbook(workbook, read_only=True, data_only=True)
    sheet = source["Learning areas"]
    wanted = {n for n, _ in table.values()}
    direct = {}
    for n, values in enumerate(
        sheet.iter_rows(min_row=min(wanted), max_row=max(wanted), values_only=True),
        min(wanted),
    ):
        if n in wanted:
            direct[n] = values
    source.close()
    for code, (n, wording) in table.items():
        row = rows[code]
        values = direct[n]
        need(
            row["source_row"] == n
            and row["attributes"]["learning_area"] == "The Arts"
            and row["attributes"]["subject"] == "Dance"
            and row["attributes"]["level"] == "Foundation Year",
            f"Wrong source level/row: {code}",
        )
        need(
            values[0] == "The Arts"
            and values[1] == "Dance"
            and values[2] == "Foundation Year"
            and values[4] == code
            and " ".join(str(values[9]).split()) == wording == row["plain_text"],
            f"Direct workbook/import/wording mismatch: {code}",
        )


def lessons_and_checks() -> None:
    lesson = (ROOT / "LESSONS.md").read_text()
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", lesson, re.MULTILINE))
    need(
        [int(m[2]) for m in marks] == list(range(31, 41)),
        "Ten daily scripts31–40 required",
    )
    for i, m in enumerate(marks):
        body = lesson[
            m.end() : marks[i + 1].start() if i + 1 < len(marks) else len(lesson)
        ]
        minutes = [
            int(n)
            for n in re.findall(
                r"^\d+\. \*\*[^\n]*?\b(\d+) min\.\*\*", body, re.MULTILINE
            )
        ]
        need(
            int(m[1]) == (int(m[2]) - 1) // 5 + 1 and minutes == [3, 4, 5, 7, 4, 2],
            "Timing/week mismatch",
        )
        need(
            "**Goal:**" in body
            and "**Prepare:**" in body
            and "**Next teaching:**" in body,
            "Daily handoff missing",
        )
        need(set(re.findall(r"\bAC9[A-Z0-9]+\b", body)) <= CODES, "Unmapped daily code")
    for name, count, pattern in [
        ("LEARNER-CARDS.md", 3, r"^- \*\*.+?:\*\* .+$"),
        (
            "OPTIONAL-PRACTICE.md",
            2,
            r"^[12]\. \*\*.+?\*\* .+?\*\*Worked report:\*\* .+$",
        ),
    ]:
        content = (ROOT / name).read_text()
        blocks = list(re.finditer(r"^## Day (\d+)\b", content, re.MULTILINE))
        need(
            [int(m[1]) for m in blocks] == list(range(31, 41)),
            "Missing day routes/examples",
        )
        for i, m in enumerate(blocks):
            body = content[
                m.end() : blocks[i + 1].start() if i + 1 < len(blocks) else len(content)
            ]
            need(
                len(re.findall(pattern, body, re.MULTILINE)) == count,
                f"Wrong routes/examples {name} day{m[1]}",
            )
    assessment = (ROOT / "ASSESSMENT.md").read_text()
    key = (ROOT / "TEACHER-KEY.md").read_text()
    ordinary = "\n".join((ROOT / name).read_text() for name in TEACHING)
    for token in (
        "MOSS BOX 47",
        "MOON PICTURE 83",
        "L → R → L → R",
        "U → V → V → FINISH",
    ):
        need(
            token in assessment and token not in ordinary,
            "Fresh source absent/leaked: " + token,
        )
    need(
        re.findall(r"teacher-held Case ([A-Z]+)", lesson)
        == re.findall(r"^## Day \d+ · fresh Case ([A-Z]+)", assessment, re.MULTILINE)
        == re.findall(r"^## Day \d+ · Case ([A-Z]+)", key, re.MULTILINE)
        == ["M", "N"],
        "Fresh IDs differ",
    )
    need(
        "first independent" in assessment
        and "spaces **1–2** and **3–4**" in key
        and "NO ACTUAL AUDIENCE" in key
        and "NOT PERFORMED" in key,
        "Assessment boundary absent",
    )
    materials = (ROOT / "MATERIALS.md").read_text()
    for token in (
        "A → B → A → B",
        "A → PAUSE → B → PAUSE → FINISH",
        "A → B → A → FINISH",
        "A → B → B → FINISH",
        "CHILD MOVEMENT OBSERVED",
        "ADULT PERFORMANCE AT CHILD DIRECTION",
        "OBJECT MODEL",
        "SCORE ONLY / NOT PERFORMED",
        "ACTUAL SOURCE ENCOUNTER",
        "ACTUAL WILLING AUDIENCE",
        "ACTUAL RESPONSE / NONE",
    ):
        need(token in materials, "Materials order/evidence field absent: " + token)
    supplied = (ROOT / "SUPPLIED-WORK.md").read_text()
    need(
        "A → B → A → B → PAUSE → FINISH" in supplied
        and "computer-created teaching model" in supplied
        and "documented author purpose" in supplied.lower()
        and "actual welcoming" not in supplied.lower(),
        "Supplied-work attribution/order drift",
    )
    need(
        "No named human choreographer" in supplied
        and "not a physical or community Dance encounter" in supplied,
        "False source identity boundary",
    )
    family = (ROOT / "FAMILY-OPTIONAL.md").read_text()
    need(
        [int(n) for n in re.findall(r"^\| (\d+) \|", family, re.MULTILINE)]
        == list(range(31, 41))
        and "not homework" in family
        and "No family needs to buy" in family,
        "Family choices absent",
    )
    aloud = (ROOT / "READ-ALOUD.md").read_text()
    need(
        len(re.findall(r"^## Track [123] ·", aloud, re.MULTILINE)) == 3
        and "not recorded audio" in aloud,
        "Narration status absent",
    )


def assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    alternatives = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text()
    font_root = Path("/usr/share/fonts/truetype/dejavu")
    for stem in STEMS:
        path = ROOT / "print" / stem
        root = ET.parse(path.with_suffix(".svg")).getroot()
        need(
            root.attrib.get("width") == "210mm"
            and root.attrib.get("height") == "297mm"
            and root.find("s:title", ns) is not None
            and root.find("s:desc", ns) is not None,
            f"A4 SVG metadata: {stem}",
        )
        info = subprocess.check_output(
            ["pdfinfo", str(path.with_suffix(".pdf"))], text=True
        )
        need("Pages:           1" in info and "(A4)" in info, f"PDF not one A4: {stem}")
        pdfwords = " ".join(
            subprocess.check_output(
                ["pdftotext", str(path.with_suffix(".pdf")), "-"], text=True
            ).split()
        )
        for node in root.findall(".//s:text", ns):
            words = "".join(node.itertext())
            need(
                " ".join(words.split()) in pdfwords and "- " + words in alternatives,
                f"Visible text/alternative mismatch: {stem}: {words}",
            )
            x, y, size = (
                float(node.attrib["x"]),
                float(node.attrib["y"]),
                float(node.attrib["font-size"]),
            )
            font = ImageFont.truetype(
                str(
                    font_root
                    / (
                        "DejaVuSans-Bold.ttf"
                        if node.attrib.get("font-weight") == "700"
                        else "DejaVuSans.ttf"
                    )
                ),
                round(size * 100),
            )
            width = font.getlength(words) / 100
            need(
                10 <= x and x + width <= 200 and size <= y <= 295,
                f"Text exceeds page safe bounds: {stem}: {words}",
            )
    with tempfile.TemporaryDirectory(prefix="subjectnest-dance-print-") as tmp:
        subprocess.run(
            ["python", str(ROOT / "print/generate_print.py"), "--output", tmp],
            check=True,
            stdout=subprocess.DEVNULL,
        )
        names = [f"{stem}.{ext}" for stem in STEMS for ext in ("svg", "pdf")] + [
            "TEXT-ALTERNATIVES.md"
        ]
        need(
            all(
                (Path(tmp) / name).read_bytes() == (ROOT / "print" / name).read_bytes()
                for name in names
            ),
            "Print reproduction differs",
        )


def interactive() -> None:
    source = (ROOT / "interactive/phrase-planner.html").read_text()
    scripts = re.findall(r"<script>([\s\S]*?)</script>", source)
    need(len(scripts) == 1, "Expected one inline script")
    script = scripts[0]
    need(
        not re.search(
            r"\b(fetch|XMLHttpRequest|localStorage|sessionStorage|indexedDB|getUserMedia|Audio|WebSocket)\b",
            script,
        ),
        "Unexpected network/storage/media API",
    )
    need(
        not re.search(
            r"(?:https?:)?//[^\s]+", script.replace("http://www.w3.org/2000/svg", "")
        ),
        "External script URL",
    )
    with tempfile.TemporaryDirectory(prefix="subjectnest-dance-js-") as tmp:
        js = Path(tmp) / "phrase.js"
        js.write_text(script)
        subprocess.run(
            ["node", "--check", str(js)], check=True, stdout=subprocess.DEVNULL
        )
    for token in (
        'id="phrase"',
        'id="pictures" type="button" aria-pressed="true"',
        'id="words" type="button" aria-pressed="false"',
        'role="status" aria-live="polite"',
        'scope="row"',
        "without JavaScript",
        "phrase.value='units';view='Pictures'",
        "original computer-created graphic teaching models",
        "not an observed performance",
    ):
        need(token in source, "Interactive control/fallback/status missing: " + token)
    for codes in (
        "['A','B','A','B','FINISH']",
        "['A','PAUSE','B','PAUSE','FINISH']",
        "['A','B','A','FINISH']",
    ):
        need(codes in script, "Interactive source order drift")
    for words in (
        "A → B → A → B → FINISH",
        "A → PAUSE → B → PAUSE → FINISH",
        "A → B → A → FINISH",
        "A → B → B → FINISH",
    ):
        need(
            words in source and words in (ROOT / "MATERIALS.md").read_text(),
            "Paper equivalence order drift",
        )


def links_and_spacing() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        content = path.read_text()
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", content):
            if url.startswith(("http://", "https://", "mailto:", "#")):
                continue
            need(
                (path.parent / unquote(url.partition("#")[0])).resolve().exists(),
                f"Broken local link {path.name}: {url}",
            )
            count += 1
        lines = content.splitlines()
        marker = re.compile(r"^(?:- |\d+\. )")
        for i, line in enumerate(lines):
            if marker.match(line):
                before = lines[i - 1] if i else ""
                after = lines[i + 1] if i + 1 < len(lines) else ""
                need(
                    not before.strip() or bool(marker.match(before)),
                    f"List blank line above: {path.name}:{i + 1}",
                )
                need(
                    not after.strip() or bool(marker.match(after)),
                    f"List blank line below: {path.name}:{i + 1}",
                )
    for url in re.findall(
        r'href="([^"]+)"', (ROOT / "interactive/phrase-planner.html").read_text()
    ):
        if not url.startswith(("http://", "https://", "#")):
            need(
                (ROOT / "interactive" / unquote(url)).resolve().exists(),
                f"Interactive local link missing: {url}",
            )
            count += 1
    return count


def manifest_data() -> dict:
    files = sorted(
        p
        for p in ROOT.rglob("*")
        if p.is_file()
        and p.name != "manifest.json"
        and not {"__pycache__", ".ruff_cache"}.intersection(p.parts)
    )
    entries = [
        {
            "item_code_or_locator": str(p.relative_to(ROOT)),
            "sha256": hashlib.sha256(p.read_bytes()).hexdigest(),
        }
        for p in files
    ]
    return {
        "pack_id": "subjectnest-acara-v9-foundation-dance-t1-w07-08",
        "pack_version": "0.1.0-draft",
        "created_at": "2026-09-30",
        "review_status": "author_and_independent_agent_review_pending_human_educator_accessibility_local_syllabus_safety_and_classroom_review",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(CODES),
        "curriculum_source": {
            "authority": "Australian Curriculum, Assessment and Reporting Authority",
            "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx",
            "retrieved_at": "2026-09-29",
            "source_hash": SOURCE_SHA,
            "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use",
        },
        "rights": "Original SubjectNest lessons, scores, drawings and text CC BY4.0; code Apache-2.0; official source/font terms retained.",
        "pack_files": entries,
        "original_assets": [
            e
            for e in entries
            if e["item_code_or_locator"]
            not in {
                "print/FONT-LICENSE.txt",
                "CODE-LICENSE.txt",
                "CURRICULUM-CROSSWALK.md",
            }
        ],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    sources()
    lessons_and_checks()
    assets()
    interactive()
    count = links_and_spacing()
    data = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(json.dumps(data, ensure_ascii=False, indent=2) + "\n")
    need(json.loads(manifest.read_text()) == data, "Hash manifest absent or stale")
    print(
        f"PASS DanceW7–8:10×25m,30routes,20worked choices,10home,2fresh checks,4direct workbook rows,8A4pairs,17reproduced files,offline planner,{count}local links,{len(data['pack_files'])}SHA files; no actual learner outcome claimed"
    )


if __name__ == "__main__":
    main()
