#!/usr/bin/env python3
"""Read-only-by-default independent audit of the Foundation Drama W5-6 pack."""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
CODES = {
    "AC9ADRFE01": "explore how and why the arts are important for people and communities",
    "AC9ADRFD01": "use play, imagination, arts knowledge, processes and/or skills to discover possibilities and develop ideas",
    "AC9ADRFC01": "create arts works that communicate ideas",
    "AC9ADRFP01": "share their arts works with audiences",
}
AIDS = (
    "aim-action",
    "obstacle-choice",
    "offer-response",
    "two-choices",
    "same-action-view",
    "turns",
    "scene-record",
)
CHECKS = ("fresh-k", "fresh-l")
REQUIRED = {
    "README.md",
    "MATERIALS.md",
    "LESSONS.md",
    "LEARNER-CARDS.md",
    "PRACTICE-SWAPS.md",
    "EXAMPLE-BANK.md",
    "PROMPTS.md",
    "STUDENT-CHECKS.md",
    "teacher/KEY-AND-NEXT.md",
    "CURRICULUM-CROSSWALK.md",
    "SOURCE-AND-RIGHTS.md",
    "AUTHENTIC-SOURCE-SLOT.md",
    "RUN-THROUGH.md",
    "print/TEXT-ALTERNATIVES.md",
    "print/FONT-RIGHTS.md",
    "print/dejavu-font-copyright.txt",
    "generate_assets.py",
    "verify_pack.py",
}
for stem in AIDS:
    REQUIRED.update((f"print/{stem}.svg", f"print/{stem}.pdf"))
for stem in CHECKS:
    REQUIRED.update((f"check/{stem}.svg", f"check/{stem}.pdf"))


def need(ok, msg):
    if not ok:
        raise AssertionError(msg)


def read(rel):
    return (ROOT / rel).read_text(encoding="utf-8")


def table_row(rel, day, n):
    found = re.findall(rf"^\| {day} \|(.+)$", read(rel), flags=re.MULTILINE)
    need(len(found) == 1, f"{rel}: Day {day} absent or duplicate")
    cells = [x.strip() for x in found[0].strip().strip("|").split("|")]
    need(len(cells) == n and all(cells), f"{rel}: Day {day} table drift")
    return cells


def check_content():
    cross = read("CURRICULUM-CROSSWALK.md")
    source = read("SOURCE-AND-RIGHTS.md")
    for code, exact in CODES.items():
        need(code in cross and exact in cross, f"exact ACARA row missing: {code}")
    for url in (
        "https://www.qcaa.qld.edu.au/downloads/aciqv9/the-arts/curriculum/ac9_drama_prep_as_cd_alignment.pdf",
        "https://www.qcaa.qld.edu.au/downloads/aciqv9/the-arts/assessment/ac9_arts_perform_tc_prep.pdf",
    ):
        need(url in cross and url in source, f"official source missing: {url}")
    need(
        "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3" in cross,
        "pinned workbook digest absent",
    )
    for phrase in ("unverified", "not a QCAA instrument", "paper devising"):
        need(phrase.lower() in cross.lower(), f"limitation absent: {phrase}")
    for phrase in ("named real maker", "cultural authority", "hold AC9ADRFE01"):
        need(
            phrase.lower() in read("AUTHENTIC-SOURCE-SLOT.md").lower(),
            f"authentic-source gate missing: {phrase}",
        )
    for phrase in ("Safe opt-in enactment gate", "paper-only", "not child performance"):
        need(
            phrase.lower() in read("MATERIALS.md").lower(),
            f"participation/evidence gate missing: {phrase}",
        )
    lessons = read("LESSONS.md")
    for day in range(21, 31):
        section = re.findall(
            rf"^## Week [56] · Day {day} ·.*?(?=^## Week |\Z)",
            lessons,
            flags=re.MULTILINE | re.DOTALL,
        )
        need(len(section) == 1, f"Day {day} lesson count")
        body = section[0]
        need("**Target/codes:**" in body and "**Prepare:**" in body, f"Day {day} setup")
        stages = (
            (
                "Welcome",
                "Access",
                "Independent plan",
                "Child response",
                "Self-check",
                "Close",
            )
            if day in (25, 30)
            else ("Welcome", "Model", "Try together", "Child choice", "Notice", "Close")
        )
        times = []
        for stage in stages:
            f = re.findall(rf"\*\*{re.escape(stage)} · (\d+) min\.\*\*", body)
            need(len(f) == 1, f"Day {day} missing/duplicate {stage}")
            times.append(int(f[0]))
        need(times == [3, 4, 5, 7, 4, 2], f"Day {day} not 25 min")
        routes = table_row("LEARNER-CARDS.md", day, 3)
        need(len(set(routes)) == 3, f"Day {day} repeated routes")
        swaps = table_row("PRACTICE-SWAPS.md", day, 2)
        need(all("→" in x for x in swaps), f"Day {day} swap not worked")
        table_row("EXAMPLE-BANK.md", day, 2)
        table_row("PROMPTS.md", day, 1)
        table_row("teacher/KEY-AND-NEXT.md", day, 2)
    student = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    need("Check K" in student and "Check L" in student, "public checks missing")
    need(
        "fresh-k.svg" in student and "fresh-l.svg" in student,
        "public check assets missing",
    )
    need(
        "Check K · worked solution" in key and "Check L · worked solution" in key,
        "separate key incomplete",
    )
    need(
        "one workable" not in student.lower()
        and "side stand"
        not in student.split("## Check K")[1].split("## Check L")[0].lower(),
        "key leaked to student check",
    )
    need(
        "After Check K" in table_row("PRACTICE-SWAPS.md", 25, 2)[0]
        and "After Check L" in table_row("PRACTICE-SWAPS.md", 30, 2)[0],
        "check practice timing unmarked",
    )
    for phrase in ("No child", "unverified", "DejaVu", "CC BY 4.0"):
        need(phrase.lower() in source.lower(), f"rights or claim absent: {phrase}")


def check_print():
    alt = read("print/TEXT-ALTERNATIVES.md")
    for folder, stems in (("print", AIDS), ("check", CHECKS)):
        for stem in stems:
            xml = ET.parse(ROOT / f"{folder}/{stem}.svg").getroot()
            need(
                xml.attrib.get("width") == "210mm"
                and xml.attrib.get("height") == "297mm"
                and xml.attrib.get("viewBox") == "0 0 794 1123",
                f"{stem} not A4 SVG",
            )
            tags = [e.tag.rsplit("}", 1)[-1] for e in xml.iter()]
            need(
                "title" in tags
                and "desc" in tags
                and "rect" in tags
                and "line" in tags,
                f"{stem} not illustrated/access named",
            )
            need(
                f"{folder}/{stem}.svg" in alt and f"{folder}/{stem}.pdf" in alt,
                f"{stem} exact alternative missing",
            )
            info = subprocess.run(
                ["pdfinfo", str(ROOT / f"{folder}/{stem}.pdf")],
                capture_output=True,
                text=True,
                check=True,
            ).stdout
            need(
                "Pages:           1" in info and "595.276 x 841.89 pts" in info,
                f"{stem} PDF not single A4",
            )
            pdftext = subprocess.run(
                ["pdftotext", str(ROOT / f"{folder}/{stem}.pdf"), "-"],
                capture_output=True,
                text=True,
                check=True,
            ).stdout
            need(
                "FOUNDATION DRAMA" in pdftext and len(pdftext.split()) > 20,
                f"{stem} PDF text unreadable",
            )
            fonts = subprocess.run(
                ["pdffonts", str(ROOT / f"{folder}/{stem}.pdf")],
                capture_output=True,
                text=True,
                check=True,
            ).stdout
            need("DejaVu" in fonts, f"{stem} font not embedded")
    need(
        "bitstream" in read("print/dejavu-font-copyright.txt").lower(),
        "font licence missing",
    )


def links():
    count = 0
    for p in ROOT.rglob("*.md"):
        s = p.read_text(encoding="utf-8")
        for target in re.findall(r"\[[^]]+\]\(([^)]+)\)", s):
            if target.startswith(("https://", "http://", "mailto:")):
                continue
            name, _, anchor = unquote(target).partition("#")
            dest = (p.parent / name).resolve() if name else p
            need(dest.is_file(), f"broken link in {p.name}: {target}")
            if anchor and dest.suffix == ".md":
                heads = re.findall(
                    r"^#{1,6} (.+)$", dest.read_text(encoding="utf-8"), re.MULTILINE
                )

                def slug(v):
                    return re.sub(
                        r"\s",
                        "-",
                        re.sub(
                            r"[^\w\s-]", "", re.sub(r"<[^>]+>", "", v).lower().strip()
                        ),
                    )

                need(anchor in {slug(h) for h in heads}, f"broken anchor {target}")
            count += 1
    return count


def hashes(write):
    files = sorted(
        p.relative_to(ROOT).as_posix()
        for p in ROOT.rglob("*")
        if p.is_file()
        and p.name != "ASSET-MANIFEST.json"
        and "__pycache__" not in p.parts
        and ".ruff_cache" not in p.parts
    )
    need(
        set(files) == REQUIRED,
        f"file set drift; missing {sorted(REQUIRED - set(files))}; extra {sorted(set(files) - REQUIRED)}",
    )
    for name in files:
        if Path(name).suffix in (".md", ".py", ".svg", ".txt"):
            s = read(name)
            need(
                s.endswith("\n")
                and "\r" not in s
                and all(line == line.rstrip() for line in s.splitlines()),
                f"{name} formatting drift",
            )
    obj = {
        "algorithm": "sha256",
        "files": {
            name: hashlib.sha256((ROOT / name).read_bytes()).hexdigest()
            for name in files
        },
    }
    if write:
        (ROOT / "ASSET-MANIFEST.json").write_text(
            json.dumps(obj, indent=2, sort_keys=True) + "\n", encoding="utf-8"
        )
    else:
        need(json.loads(read("ASSET-MANIFEST.json")) == obj, "manifest drift")
    return len(files)


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--write-manifest", action="store_true")
    a = ap.parse_args()
    check_content()
    check_print()
    n = links()
    f = hashes(a.write_manifest)
    print(
        f"PASS: 10 x 25-minute scripts; 30 routes; 20 worked swaps; 10 bridges; 2 independent checks; 9 A4 SVG/PDF pairs; {n} local links; {f} SHA256 files"
    )


if __name__ == "__main__":
    main()
