#!/usr/bin/env python3
"""Read-only default integrity, source, content, link and asset audit."""

from __future__ import annotations

import argparse
import ast
import hashlib
import json
import re
import subprocess
import tempfile
import warnings
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

from openpyxl import load_workbook

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
ROWS = {"AC9TDIFK01": 20098, "AC9TDIFK02": 20105, "AC9TDIFP01": 20111}
STEMS = (
    "message-board",
    "studio-key-1",
    "studio-key-2",
    "reader-repair",
    "personal-category-mat",
    "revision-record",
    "check-a-card-show",
    "check-b-two-keys",
)


def need(ok, message):
    if not ok:
        raise AssertionError(message)


def read(name):
    return (ROOT / name).read_text(encoding="utf-8")


def digest(path):
    return hashlib.sha256(path.read_bytes()).hexdigest()


def normal(value):
    return re.sub(r"\s+", "", value)


def sources():
    path = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    source = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text())
    snapshot = json.loads(read("source-snapshot.json"))
    crosswalk = read("CURRICULUM-CROSSWALK.md")
    need(
        digest(path)
        == source["source_sha256"]
        == snapshot["live_workbook_sha256"]
        == SHA,
        "workbook/import/live receipt SHA drift",
    )
    need(
        snapshot["http_status"] == 200 and snapshot["content_rows"] == ROWS,
        "source receipt/row drift",
    )
    records = {
        r["code"]: r
        for r in source["records"]
        if r.get("record_type") == "content_description" and r.get("code") in ROWS
    }
    with warnings.catch_warnings():
        warnings.simplefilter("ignore", UserWarning)
        book = load_workbook(path, read_only=True, data_only=True)
    wanted = set(ROWS.values())
    actual = {}
    for number, row in enumerate(book["Learning areas"].iter_rows(values_only=True), 1):
        if number in wanted:
            actual[number] = row
        if number > max(wanted):
            break
    book.close()
    for code, rownum in ROWS.items():
        record = records[code]
        attrs = record["attributes"]
        row = actual[rownum]
        need(
            record["source_row"] == rownum
            and attrs["level"] == "Foundation Year"
            and attrs["subject"] == "Digital Technologies",
            "source placement drift " + code,
        )
        need(
            row[4] == code
            and row[0] == "Technologies"
            and row[1] == "Digital Technologies"
            and row[2] == "Foundation Year",
            "direct workbook row drift " + code,
        )
        need(
            normal(str(row[9])) == normal(record["plain_text"]),
            "direct/import wording drift " + code,
        )
        need(
            f"| {code} · {rownum} | {record['plain_text']} |" in crosswalk,
            "crosswalk exact wording drift " + code,
        )
    need(
        snapshot["official_workbook_url"] in crosswalk
        and snapshot["qcaa_alignment_url"] in crosswalk,
        "source URL missing",
    )
    need(
        "AC9TDI2K02" in crosswalk
        and "Direct PDF access was blocked" in crosswalk
        and "Years 1–2" in crosswalk,
        "source discrepancy/limits missing",
    )
    print(
        "PASS official SHA and three exact direct workbook/import/crosswalk rows; honest QCAA access limit"
    )


def curriculum_content():
    lessons = read("LESSONS.md")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", lessons, re.MULTILINE))
    need([int(m[2]) for m in marks] == list(range(31, 41)), "daily sequence incomplete")
    bodies = []
    for i, m in enumerate(marks):
        body = lessons[
            m.end() : marks[i + 1].start() if i + 1 < len(marks) else len(lessons)
        ]
        bodies.append(body)
        need(int(m[1]) == (int(m[2]) - 1) // 5 + 1, "day/week mismatch")
        minutes = [
            int(v)
            for v in re.findall(
                r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*", body, re.MULTILINE
            )
        ]
        need(minutes == [3, 4, 5, 7, 4, 2], "timing drift day " + m[2])
        need(
            "**Child task · 7 min.**" in body and any(code in body for code in ROWS),
            "task/code missing day " + m[2],
        )
    need(len(set(bodies)) == 10, "duplicate daily scripts")
    cards = read("LEARNER-CARDS.md")
    swaps = read("PRACTICE-SWAPS.md")
    need(
        len(re.findall(r"^\| (?:3[1-9]|40) ·", cards, re.MULTILINE)) == 10,
        "30 routes / ten bridges incomplete",
    )
    need(
        len(re.findall(r"^\| (?:3[1-9]|40)(?: · AFTER)? \|", swaps, re.MULTILINE))
        == 10,
        "twenty worked variations incomplete",
    )
    for day in (35, 40):
        need(f"| {day} · AFTER |" in swaps, "check-day practice before capture")
    check = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    need(
        "SHOW RING → LIFT FLAG → FINISH" in check
        and "CLOSE CARD → OPEN CARD → SHOW SUN" in check,
        "fresh source drift",
    )
    need(
        "RECTANGLE → OVAL → DIAMOND" in key
        and "TRIANGLE → SQUARE → CIRCLE" in key
        and "TRIANGLE → CIRCLE → SQUARE" in key,
        "fresh key drift",
    )
    need(
        "RECTANGLE → OVAL → DIAMOND" not in check
        and "TRIANGLE → CIRCLE → SQUARE" not in check,
        "answer leaked to learner copy",
    )
    need(
        "NOT OBSERVED" in read("DEVICE-GATE.md") and "actual" in read("DEVICE-GATE.md"),
        "actual device evidence gate missing",
    )
    print(
        "PASS ten distinct 25-minute scripts, 30 routes, 20 worked swaps, ten bridges, fresh checks/separate key"
    )


def print_assets():
    alt = read("print/TEXT-ALTERNATIVES.md")
    for stem in STEMS:
        svg = ROOT / "print" / (stem + ".svg")
        pdf = svg.with_suffix(".pdf")
        tree = ET.fromstring(svg.read_text())
        need(
            tree.attrib.get("width") == "210mm"
            and tree.attrib.get("height") == "297mm"
            and tree.attrib.get("viewBox") == "0 0 794 1123",
            "A4 SVG drift " + stem,
        )
        need(stem in alt, "alternative missing " + stem)
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        need(
            re.search(r"Pages:\s+1\b", info)
            and re.search(r"Page size:\s+595\.\d+ x 841\.\d+ pts", info),
            "A4 PDF drift " + stem,
        )
        pdftext = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        for node in tree.iter():
            if node.tag.endswith("}text"):
                value = node.text or ""
                need(
                    value in alt and normal(value) in normal(pdftext),
                    "visible text/alternative/PDF mismatch " + stem + ": " + value,
                )
    with tempfile.TemporaryDirectory(prefix="subjectnest-message-print-") as temp:
        out = Path(temp)
        subprocess.run(
            ["python", str(ROOT / "print/generate_print.py"), "--output", temp],
            check=True,
            stdout=subprocess.DEVNULL,
        )
        for file in out.iterdir():
            need(
                digest(file) == digest(ROOT / "print" / file.name),
                "non-reproducible asset " + file.name,
            )
    print(
        "PASS eight A4 SVG/searchable PDF pairs, complete exact alternatives, and 17 byte-reproduced outputs"
    )


def code_and_links():
    for path in ROOT.rglob("*.py"):
        ast.parse(path.read_text())
    html = read("interactive/message-studio.html")
    scripts = re.findall(r"<script>(.*?)</script>", html, re.DOTALL)
    with tempfile.TemporaryDirectory(prefix="subjectnest-message-js-") as temp:
        path = Path(temp) / "app.js"
        path.write_text("\n".join(scripts))
        subprocess.run(
            ["node", "--check", str(path)], check=True, stdout=subprocess.DEVNULL
        )
    need(
        not re.search(
            r"fetch\s*\(|XMLHttpRequest|localStorage|sessionStorage|getUserMedia|serviceWorker|<script[^>]+src",
            html,
        ),
        "unexpected data/network/device API",
    )
    need(
        'role="status"' in html
        and "aria-pressed" in html
        and '<label for="message">' in html
        and '<label for="key">' in html,
        "label/state controls missing",
    )
    need(
        "SHOW RING" not in html and "Desk Key R" not in html,
        "fresh check source exposed in interactive",
    )
    count = 0
    for file in ROOT.rglob("*.md"):
        for target in re.findall(r"\]\(([^)]+)\)", file.read_text()):
            if re.match(r"[a-z]+:|#", target):
                continue
            route, _, anchor = unquote(target).partition("#")
            resolved = (file.parent / route).resolve()
            need(
                resolved.exists(),
                "broken link " + str(file.relative_to(ROOT)) + ": " + target,
            )
            if anchor and resolved.suffix == ".md":
                headings = re.findall(
                    r"^#{1,6} (.+)$", resolved.read_text(), re.MULTILINE
                )
                slugs = {
                    re.sub(r"[^\w\- ]", "", h.lower()).replace(" ", "-")
                    for h in headings
                }
                need(anchor in slugs, "broken anchor " + target)
            count += 1
    for target in re.findall(r'href="([^"#]+)"', html):
        need(
            (ROOT / "interactive" / target).resolve().exists(),
            "interactive companion missing " + target,
        )
    print(
        f"PASS JS syntax, explicit labels/fixed local data/complete paper route and {count} local links"
    )


def inventory():
    return {
        str(p.relative_to(ROOT)): digest(p)
        for p in sorted(ROOT.rglob("*"))
        if p.is_file() and p.name != "manifest.json" and "__pycache__" not in p.parts
    }


def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    sources()
    curriculum_content()
    print_assets()
    code_and_links()
    files = inventory()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(
            json.dumps(
                {
                    "schema": "subjectnest.pack.v1",
                    "stage": "authored; human/local review pending",
                    "days": list(range(31, 41)),
                    "source_workbook_sha256": SHA,
                    "files": files,
                },
                indent=2,
            )
            + "\n"
        )
    else:
        need(
            json.loads(manifest.read_text())["files"] == files,
            "exact file inventory/SHA drift",
        )
    print(
        f"PASS {len(files)} SHA-256 files · author desk QA only; no classroom/child study"
    )


if __name__ == "__main__":
    main()
