#!/usr/bin/env python3
"""Read-only structural, source, print and hash audit; --write-manifest is explicit."""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import sys
import unicodedata
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

from write_crosswalk import render

ROOT = Path(__file__).resolve().parent
AREAS = ("science", "hass", "hpe", "arts")
DAYS = tuple(range(11, 21))
PRINTS = ("science-sky", "hass-source", "hpe-choice", "arts-workbench")
LINK = re.compile(r"(?<!!)\[[^\]]+\]\(([^)]+)\)")
DAY = re.compile(r"(?m)^## Day (\d+)\b")


def sha(path: Path) -> str:
    return hashlib.sha256(path.read_bytes()).hexdigest()


def files() -> list[Path]:
    return sorted(p for p in ROOT.rglob("*") if p.is_file() and p.name != "MANIFEST.sha256" and "__pycache__" not in p.parts)


def slug(heading: str) -> str:
    heading = re.sub(r"<[^>]*>", "", heading).strip().lower()
    heading = unicodedata.normalize("NFKD", heading)
    heading = "".join(c for c in heading if c.isalnum() or c in " _-")
    return re.sub(r" +", "-", heading)


def main() -> int:
    ap = argparse.ArgumentParser(description=__doc__)
    ap.add_argument("--write-manifest", action="store_true", help="regenerate hashes after reviewing edits")
    args = ap.parse_args()
    errors: list[str] = []

    def check(ok: bool, why: str) -> None:
        if not ok:
            errors.append(why)

    required = ["README.md", "CURRICULUM-CROSSWALK.md", "SOURCES-AND-REVIEW.md", "RUN-THROUGH.md", "CODE-LICENSE.txt", "curriculum_codes.json", "write_crosswalk.py", "verify_pack.py", "print/TEXT-ALTERNATIVES.md", "print/generate_print.py", "print/dejavu-font-copyright.txt"]
    required += [f"{area}/{part}.md" for area in AREAS for part in ("TEACHER", "LEARNER", "ASSESSMENT")]
    required += [f"print/{name}.{ext}" for name in PRINTS for ext in ("svg", "pdf")]
    for name in required:
        check((ROOT / name).is_file(), f"missing {name}")
    check(not list((ROOT / "print").glob("*-preview.png")), "temporary preview PNG left in pack")

    for area in AREAS:
        teacher = (ROOT / area / "TEACHER.md").read_text(encoding="utf-8")
        learner = (ROOT / area / "LEARNER.md").read_text(encoding="utf-8")
        staff = (ROOT / area / "ASSESSMENT.md").read_text(encoding="utf-8")
        for name, content in (("teacher", teacher), ("learner", learner)):
            found = [int(x) for x in DAY.findall(content)]
            check(found == list(DAYS), f"{area} {name}: expected Days 11–20 once, got {found}")
        check("ASSESSMENT.md" not in learner and "staff key" not in learner.lower(), f"{area} learner links/leaks staff key")
        chunks = re.split(r"(?m)^## Day \d+\b[^\n]*\n", teacher)[1:]
        check(len(chunks) == 10, f"{area} has {len(chunks)} teacher blocks")
        for day, chunk in zip(DAYS, chunks):
            for pattern, label in ((r"\*\*Setup:\*\*", "setup"), (r"(?m)^1\. \*\*[^\n]*2 min", "notice 2"), (r"(?m)^2\. \*\*[^\n]*7 min", "try 7"), (r"(?m)^3\. \*\*[^\n]*3 min", "show 3"), (r"\*\*Access/home:\*\*", "access/home"), (r"\*\*(?:Limit|Evidence boundary):\*\*", "evidence limit")):
                check(bool(re.search(pattern, chunk)), f"{area} Day {day} missing {label}")
            check(bool(re.search(r"\*\*Codes:\*\*", chunk)), f"{area} Day {day} missing codes")
        for day in DAYS:
            check(bool(re.search(rf"(?m)^\| {day}(?: \|| [A-Za-z])", staff)), f"{area} Day {day} missing staff evidence/next move")

    arts = (ROOT / "arts/TEACHER.md").read_text(encoding="utf-8")
    for form in ("Dance", "Drama", "Media Arts", "Music", "Visual Arts"):
        check(len(re.findall(rf"(?m)^## Day \d+ — {re.escape(form)}:", arts)) == 2, f"{form} does not have exactly two blocks")

    crosswalk = (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8")
    try:
        check(crosswalk == render(), "crosswalk differs from pinned ACARA import")
    except Exception as exc:
        errors.append(f"ACARA crosswalk verification failed: {exc}")
    selected = json.loads((ROOT / "curriculum_codes.json").read_text(encoding="utf-8"))
    for area, codes in selected.items():
        for code in codes:
            check(code in (ROOT / area.lower() / "TEACHER.md").read_text(encoding="utf-8"), f"{code} absent from {area} teacher sequence")
            check(f"`{code}`" in crosswalk, f"{code} absent from crosswalk")
    check(sum(map(len, selected.values())) == 39, "selected code count changed; review coverage statement")
    check("AC9HS2K01" not in selected["HASS"] and "AC9HS2K04" not in selected["HASS"], "excluded local/Country codes selected")

    for md in ROOT.rglob("*.md"):
        content = md.read_text(encoding="utf-8")
        for raw in LINK.findall(content):
            if raw.startswith(("https://", "http://", "mailto:", "data:")):
                continue
            loc, _, fragment = unquote(raw).partition("#")
            dest = (md.parent / loc).resolve() if loc else md
            if not (args.write_manifest and dest == ROOT / "MANIFEST.sha256"):
                check(dest.is_file(), f"broken local link {md.relative_to(ROOT)} -> {raw}")
            if fragment and dest.suffix == ".md" and dest.is_file():
                text = dest.read_text(encoding="utf-8")
                anchors = set(re.findall(r'<a id="([^"]+)"', text))
                anchors |= {slug(h) for h in re.findall(r"(?m)^#{1,6} +(.+)$", text)}
                check(fragment in anchors, f"broken anchor {md.relative_to(ROOT)} -> {raw}")

    text_alt = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text(encoding="utf-8")
    for name in PRINTS:
        svg = ROOT / "print" / f"{name}.svg"
        pdf = ROOT / "print" / f"{name}.pdf"
        if svg.exists():
            raw = svg.read_text(encoding="utf-8")
            check('width="210mm"' in raw and 'height="297mm"' in raw, f"{name} SVG is not A4")
            try:
                root = ET.fromstring(raw)
                check(root.attrib.get("role") == "img", f"{name} SVG lacks img role")
                check(any(x.tag.endswith("title") for x in root), f"{name} SVG lacks title")
                check(any(x.tag.endswith("desc") for x in root), f"{name} SVG lacks description")
            except ET.ParseError as exc:
                errors.append(f"{name} SVG invalid: {exc}")
        if pdf.exists():
            result = subprocess.run(["pdfinfo", str(pdf)], text=True, capture_output=True)
            check(result.returncode == 0 and "Pages:           1" in result.stdout and "(A4)" in result.stdout, f"{name} PDF not single A4 page")
        alt_headings = {slug(h) for h in re.findall(r"(?m)^#{1,6} +(.+)$", text_alt)}
        check(f"{name}-aid" in alt_headings, f"{name} missing specific text alternative")

    if args.write_manifest:
        if errors:
            print("\n".join(errors), file=sys.stderr)
            return 1
        lines = [f"{sha(p)}  {p.relative_to(ROOT)}" for p in files()]
        (ROOT / "MANIFEST.sha256").write_text("\n".join(lines) + "\n", encoding="utf-8")
        print(f"Wrote {len(lines)} hashes")
    else:
        manifest = ROOT / "MANIFEST.sha256"
        if not manifest.exists():
            errors.append("missing MANIFEST.sha256; run --write-manifest after reviewing")
        else:
            actual = {str(p.relative_to(ROOT)): sha(p) for p in files()}
            expected = {}
            for line in manifest.read_text(encoding="utf-8").splitlines():
                match = re.fullmatch(r"([0-9a-f]{64})  (.+)", line)
                if not match:
                    errors.append(f"malformed manifest line: {line[:60]}")
                else:
                    expected[match[2]] = match[1]
            check(expected == actual, "manifest mismatches or omits files")

    if errors:
        print("FAILED:\n- " + "\n- ".join(errors), file=sys.stderr)
        return 1
    print("PASS: 40 distinct day blocks; staff/learner split; 39 exact band codes; local links; four A4 pairs/text alternatives; hashes")
    return 0


if __name__ == "__main__":
    raise SystemExit(main())
