#!/usr/bin/env python3
"""Read-only-by-default content, source, print and digest audit for Dance W5-6."""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
CODES = {
    "AC9ADAFE01": (
        "20513",
        "explore how and why the arts are important for people and communities",
    ),
    "AC9ADAFD01": (
        "20522",
        "use play, imagination, arts knowledge, processes and/or skills to discover possibilities and develop ideas",
    ),
    "AC9ADAFC01": ("20532", "create arts works that communicate ideas"),
    "AC9ADAFP01": ("20541", "share their arts works with audiences"),
}
AIDS = (
    "one-mark-two-sizes",
    "size-order",
    "relative-bands",
    "motif-new-position",
    "size-or-band",
    "three-moment-phrase",
    "score-and-evidence",
)
CHECKS = ("fresh-k", "fresh-l")
REQUIRED = {
    "README.md",
    "MATERIALS.md",
    "LESSONS.md",
    "LEARNER-CARDS.md",
    "PRACTICE-SWAPS.md",
    "EXAMPLE-BANK.md",
    "PROMPTS.md",
    "STUDENT-CHECKS.md",
    "teacher/KEY-AND-NEXT.md",
    "CURRICULUM-CROSSWALK.md",
    "SOURCE-AND-RIGHTS.md",
    "AUTHENTIC-SOURCE-SLOT.md",
    "RUN-THROUGH.md",
    "print/TEXT-ALTERNATIVES.md",
    "print/FONT-RIGHTS.md",
    "print/dejavu-font-copyright.txt",
    "generate_assets.py",
    "verify_pack.py",
}
for stem in AIDS:
    REQUIRED.update((f"print/{stem}.svg", f"print/{stem}.pdf"))
for stem in CHECKS:
    REQUIRED.update((f"check/{stem}.svg", f"check/{stem}.pdf"))


def need(ok, msg):
    if not ok:
        raise AssertionError(msg)


def read(rel):
    return (ROOT / rel).read_text(encoding="utf-8")


def row(rel, day, n):
    found = re.findall(rf"^\| {day} \|(.+)$", read(rel), re.MULTILINE)
    need(len(found) == 1, f"{rel}: Day {day} missing or duplicate")
    cells = [x.strip() for x in found[0].strip().strip("|").split("|")]
    need(len(cells) == n and all(cells), f"{rel}: Day {day} cells missing")
    return cells


def content():
    cross = read("CURRICULUM-CROSSWALK.md")
    rights = read("SOURCE-AND-RIGHTS.md")
    for code, (source_row, exact) in CODES.items():
        need(
            code in cross and source_row in cross and exact in cross,
            f"exact official row {code} absent",
        )
    for url in (
        "https://www.qcaa.qld.edu.au/downloads/aciqv9/the-arts/curriculum/ac9_dance_prep_as_cd_alignment.pdf",
        "https://www.qcaa.qld.edu.au/downloads/aciqv9/the-arts/assessment/ac9_arts_perform_tc_prep.pdf",
    ):
        need(url in cross and url in rights, f"QCAA official pin missing: {url}")
    need(
        "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3" in cross,
        "ACARA workbook pin absent",
    )
    for word in ("unverified", "not QCAA instruments", "paper score"):
        need(word.lower() in cross.lower(), f"curriculum limit missing: {word}")
    slot = read("AUTHENTIC-SOURCE-SLOT.md").lower()
    for word in ("named maker", "cultural authority", "hold ac9adafe01"):
        need(word in slot, f"authentic source gate missing: {word}")
    materials = read("MATERIALS.md").lower()
    for word in (
        "safety gate",
        "source a",
        "source b",
        "source c",
        "source d",
        "not observed",
    ):
        need(word in materials, f"materials source or gate missing: {word}")
    for word in ("dejavu", "cc by 4.0", "no pilot", "none embedded"):
        need(word in rights.lower(), f"rights limit missing: {word}")
    lessons = read("LESSONS.md")
    for day in range(21, 31):
        section = re.findall(
            rf"^## Week [56] · Day {day} ·.*?(?=^## Week |\Z)",
            lessons,
            re.MULTILINE | re.DOTALL,
        )
        need(len(section) == 1, f"Day {day} lesson count")
        b = section[0]
        need("**Goal/codes:**" in b and "**Prepare:**" in b, f"Day {day} setup absent")
        stages = (
            (
                "Welcome",
                "Source access",
                "Process only",
                "First independent response",
                "Self-check",
                "Close",
            )
            if day in (25, 30)
            else (
                "Welcome",
                "Read and model",
                "Try together",
                "Child choice",
                "Notice",
                "Close",
            )
        )
        times = []
        for stage in stages:
            x = re.findall(rf"\*\*{re.escape(stage)} · (\d+) min\.\*\*", b)
            need(len(x) == 1, f"Day {day} {stage} absent/duplicate")
            times.append(int(x[0]))
        need(times == [3, 4, 5, 7, 4, 2], f"Day {day} not 25 minutes")
        routes = row("LEARNER-CARDS.md", day, 3)
        need(len(set(routes)) == 3, f"Day {day} repeated route")
        swaps = row("PRACTICE-SWAPS.md", day, 2)
        need(all("→" in x for x in swaps), f"Day {day} not worked swap")
        row("EXAMPLE-BANK.md", day, 2)
        row("PROMPTS.md", day, 1)
        row("teacher/KEY-AND-NEXT.md", day, 2)
    student = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    need(
        "Check K" in student
        and "Check L" in student
        and "fresh-k.pdf" in student
        and "fresh-l.pdf" in student,
        "public check missing",
    )
    need(
        "Check K · one workable score" in key and "Check L · one workable score" in key,
        "separate key missing",
    )
    need(
        "somewhat broader" not in student
        and "compact hook in LOWER → moment 2" not in student,
        "answer leaked",
    )
    need(
        "After Check K" in row("PRACTICE-SWAPS.md", 25, 2)[0]
        and "After Check L" in row("PRACTICE-SWAPS.md", 30, 2)[0],
        "practice timing missing",
    )


def print_assets():
    alt = read("print/TEXT-ALTERNATIVES.md")
    for folder, stems in (("print", AIDS), ("check", CHECKS)):
        for stem in stems:
            root = ET.parse(ROOT / f"{folder}/{stem}.svg").getroot()
            need(
                root.attrib.get("width") == "210mm"
                and root.attrib.get("height") == "297mm"
                and root.attrib.get("viewBox") == "0 0 794 1123",
                f"{stem} SVG not A4",
            )
            tags = [e.tag.rsplit("}", 1)[-1] for e in root.iter()]
            need(
                "title" in tags
                and "desc" in tags
                and ("path" in tags or "polygon" in tags),
                f"{stem} not titled/illustrated",
            )
            need(
                f"{folder}/{stem}.svg" in alt and f"{folder}/{stem}.pdf" in alt,
                f"{stem} exact alternative absent",
            )
            pdf = ROOT / f"{folder}/{stem}.pdf"
            info = subprocess.run(
                ["pdfinfo", str(pdf)], check=True, capture_output=True, text=True
            ).stdout
            need(
                "Pages:           1" in info and "595.276 x 841.89 pts" in info,
                f"{stem} PDF not single A4",
            )
            txt = subprocess.run(
                ["pdftotext", str(pdf), "-"], check=True, capture_output=True, text=True
            ).stdout
            need(
                "FOUNDATION DANCE" in txt and len(txt.split()) > 15,
                f"{stem} PDF text absent",
            )
            fonts = subprocess.run(
                ["pdffonts", str(pdf)], check=True, capture_output=True, text=True
            ).stdout
            need("DejaVu" in fonts, f"{stem} font not embedded")
    need(
        "bitstream" in read("print/dejavu-font-copyright.txt").lower(),
        "font rights missing",
    )


def slug(v):
    return re.sub(
        r"\s", "-", re.sub(r"[^\w\s-]", "", re.sub(r"<[^>]+>", "", v).lower().strip())
    )


def links():
    n = 0
    for p in ROOT.rglob("*.md"):
        s = p.read_text(encoding="utf-8")
        for target in re.findall(r"\[[^]]+\]\(([^)]+)\)", s):
            if target.startswith(("http://", "https://", "mailto:")):
                continue
            name, _, anchor = unquote(target).partition("#")
            dest = (p.parent / name).resolve() if name else p
            need(dest.is_file(), f"broken link in {p.name}: {target}")
            if anchor and dest.suffix == ".md":
                heads = re.findall(
                    r"^#{1,6} (.+)$", dest.read_text(encoding="utf-8"), re.MULTILINE
                )
                need(anchor in {slug(h) for h in heads}, f"broken anchor {target}")
            n += 1
    return n


def files_hashes(write):
    files = sorted(
        p.relative_to(ROOT).as_posix()
        for p in ROOT.rglob("*")
        if p.is_file()
        and p.name != "ASSET-MANIFEST.json"
        and "__pycache__" not in p.parts
        and ".ruff_cache" not in p.parts
    )
    need(
        set(files) == REQUIRED,
        f"file set drift missing={sorted(REQUIRED - set(files))} extra={sorted(set(files) - REQUIRED)}",
    )
    for name in files:
        if Path(name).suffix in (".md", ".py", ".svg", ".txt"):
            s = read(name)
            need(
                s.endswith("\n")
                and "\r" not in s
                and all(line == line.rstrip() for line in s.splitlines()),
                f"{name} whitespace",
            )
    obj = {
        "algorithm": "sha256",
        "files": {
            name: hashlib.sha256((ROOT / name).read_bytes()).hexdigest()
            for name in files
        },
    }
    if write:
        (ROOT / "ASSET-MANIFEST.json").write_text(
            json.dumps(obj, indent=2, sort_keys=True) + "\n", encoding="utf-8"
        )
    else:
        need(json.loads(read("ASSET-MANIFEST.json")) == obj, "manifest drift")
    return len(files)


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--write-manifest", action="store_true")
    a = ap.parse_args()
    content()
    print_assets()
    n = links()
    f = files_hashes(a.write_manifest)
    print(
        f"PASS: 10 x 25 scripts; 30 access routes; 20 worked swaps; 10 optional bridges; 2 fresh separate checks; 9 A4 SVG/PDF pairs; {n} local links; {f} SHA256 files"
    )


if __name__ == "__main__":
    main()
