# SPDX-License-Identifier: Apache-2.0
"""Read-only curriculum, pedagogy, source-scope, print and hash audit for W27–28."""
from __future__ import annotations

import argparse
import ast
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote
from zipfile import ZipFile

from PIL import ImageFont

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
ROWS = {
    "AC9HSFK01": 1213,
    "AC9HSFK02": 1217,
    "AC9HSFK03": 1221,
    "AC9HSFK04": 1226,
    "AC9HSFS01": 1233,
    "AC9HSFS02": 1237,
    "AC9HSFS03": 1241,
    "AC9HSFS04": 1245,
    "AC9HSFS05": 1251,
}
HELD = {f"AC9HSFK0{i}" for i in range(1, 5)}
SKILLS = set(ROWS) - HELD
QCAA = "https://www.qcaa.qld.edu.au/p-10/aciq/version-9/learning-areas/p-10-humanities-and-social-sciences/hass"
ALIGN = "https://www.qcaa.qld.edu.au/downloads/aciqv9/humanities-and-social-sciences/curriculum/ac9_hass_prep_as_cd_alignment.pdf"
STEMS = (
    "paper-marker-archive-card",
    "mira-maker-note",
    "draft-caption",
    "mira-correction",
    "source-scope-mat",
    "caption-revision-mat",
    "check-a-larch-box",
    "check-b-cove-shelf",
)
EXPECTED = {
    131: {"AC9HSFS01", "AC9HSFS02", "AC9HSFS04"},
    132: {"AC9HSFS01", "AC9HSFS02", "AC9HSFS04"},
    133: {"AC9HSFS03", "AC9HSFS04"},
    134: SKILLS,
    135: SKILLS,
    136: {"AC9HSFS01", "AC9HSFS02", "AC9HSFS04"},
    137: {"AC9HSFS01", "AC9HSFS03", "AC9HSFS04"},
    138: {"AC9HSFS02", "AC9HSFS03", "AC9HSFS04"},
    139: SKILLS,
    140: SKILLS,
}
REQUIRED = {
    "README.md",
    "LESSONS.md",
    "MATERIALS.md",
    "LEARNER-CARDS.md",
    "PRACTICE-SWAPS.md",
    "FAMILY-CARER-OPTIONS.md",
    "STUDENT-CHECKS.md",
    "teacher/KEY-AND-NEXT.md",
    "CURRICULUM-CROSSWALK.md",
    "LOCAL-SOURCE-INSERT.md",
    "SOURCE-AND-RIGHTS.md",
    "RUN-THROUGH.md",
    "source-snapshot.json",
    "CODE-LICENSE.txt",
    "verify_pack.py",
    "print/generate_print.py",
    "print/TEXT-ALTERNATIVES.md",
    "print/DEJAVU-FONT-LICENSE.txt",
}
REQUIRED.update(f"print/{s}.{e}" for s in STEMS for e in ("svg", "pdf"))
NS = {"s": "http://www.w3.org/2000/svg"}
XNS = {"x": "http://schemas.openxmlformats.org/spreadsheetml/2006/main"}
FONT_ROOT = Path("/usr/share/fonts/truetype/dejavu")


def need(ok: object, message: str) -> None:
    """Stop at a concrete missing or drifted requirement."""
    if not ok:
        raise AssertionError(message)


def read(name: str) -> str:
    """Read a named pack text file without writing."""
    return (ROOT / name).read_text(encoding="utf-8")


def compact(value: str) -> str:
    """Normalise layout whitespace while preserving exact source words."""
    return " ".join(value.split())


def table_row(name: str, day: int, count: int) -> list[str]:
    """Require one nonempty fixed-width participation row per day."""
    rows = re.findall(rf"^\| {day} \|(.+)$", read(name), re.MULTILINE)
    need(len(rows) == 1, f"{name}: Day {day} absent/duplicated")
    cells = [x.strip() for x in rows[0].strip().strip("|").split("|")]
    need(len(cells) == count and all(cells), f"{name}: Day {day} cell drift")
    return cells


def workbook_rows(workbook: Path) -> dict[int, list[str]]:
    """Read the actual pinned Learning areas worksheet using standard XML."""
    with ZipFile(workbook) as z:
        strings: list[str] = []
        if "xl/sharedStrings.xml" in z.namelist():
            strings = [
                "".join(si.itertext())
                for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall(
                    "x:si", XNS
                )
            ]
        sheet = ET.fromstring(z.read("xl/worksheets/sheet1.xml"))
        out: dict[int, list[str]] = {}
        for row in sheet.findall("x:sheetData/x:row", XNS):
            number = int(row.attrib["r"])
            if number not in ROWS.values():
                continue
            cells = [""] * 12
            for cell in row.findall("x:c", XNS):
                letters = re.match(r"[A-Z]+", cell.attrib["r"]).group(0)
                col = 0
                for char in letters:
                    col = col * 26 + ord(char) - 64
                val = cell.find("x:v", XNS)
                content = val.text if val is not None and val.text else ""
                if cell.get("t") == "s":
                    content = strings[int(content)]
                elif cell.get("t") == "inlineStr":
                    content = "".join(cell.find("x:is", XNS).itertext())
                if col <= 12:
                    cells[col - 1] = compact(content)
            out[number] = cells
        return out


def sources() -> None:
    """Compare direct workbook rows, imported records and the exact crosswalk."""
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    imported = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text())
    snapshot = json.loads(read("source-snapshot.json"))
    cross = read("CURRICULUM-CROSSWALK.md")
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == imported["source_sha256"]
        == snapshot["workbook_sha256"]
        == SHA,
        "official workbook SHA drift",
    )
    need(snapshot["content_rows"] == ROWS, "official row pins drift")
    need(
        snapshot["queensland_prep_entry_point_url"] == QCAA
        and snapshot["queensland_prep_alignment_url"] == ALIGN
        and QCAA in cross
        and ALIGN in cross,
        "QCAA pins drift",
    )
    need(
        set(snapshot["skill_practice_partial_codes"]) == SKILLS
        and not snapshot["fictional_concept_rehearsal_only_codes"]
        and set(snapshot["knowledge_codes_held_for_authentic_local_sources"]) == HELD,
        "skill/knowledge boundary drift",
    )
    records = {
        r["code"]: r
        for r in imported["records"]
        if r.get("record_type") == "content_description" and r.get("code") in ROWS
    }
    direct = workbook_rows(workbook)
    need(
        set(records) == set(ROWS) and set(direct) == set(ROWS.values()),
        "official Foundation set absent",
    )
    for code, number in ROWS.items():
        r = records[code]
        values = direct[number]
        need(
            values[:3]
            == ["Humanities and Social Sciences", "HASS F-6", "Foundation Year"]
            and values[4] == code
            and values[9] == compact(r["plain_text"])
            and r["source_row"] == number,
            f"direct source row drift {code}",
        )
        pattern = rf"^\| {code} \| {number} \| Humanities and Social Sciences · HASS F-6 · Foundation Year \| {re.escape(values[9])} \|"
        need(re.search(pattern, cross, re.MULTILINE), f"exact crosswalk drift {code}")
        if code in HELD:
            need(
                "hold"
                in next(
                    line
                    for line in cross.splitlines()
                    if line.startswith(f"| {code} |")
                ).lower(),
                f"authentic hold absent {code}",
            )
    need(
        snapshot["live_verification"]["checked_at"] == "2026-09-30"
        and snapshot["live_verification"]["acara_workbook"]["sha256"] == SHA,
        "current author source receipt absent",
    )
    for term in ("other states", "not secure exams", "no k code", "temporal terms"):
        need(term in cross.lower(), f"crosswalk boundary absent: {term}")
    gate = read("LOCAL-SOURCE-INSERT.md").lower()
    need(
        all(
            t in gate
            for t in (
                "cultural authority",
                "hold the relevant knowledge claim",
                "icip",
                "free, prior and informed consent",
            )
        ),
        "local/ICIP gate absent",
    )


def pedagogy() -> None:
    """Verify timed distinct lessons and same-goal routes/transfers/bridges."""
    lessons = read("LESSONS.md")
    marks = list(
        re.finditer(r"^## Week (\d+) · Day (\d+) · (.+)$", lessons, re.MULTILINE)
    )
    need(
        [int(m.group(2)) for m in marks] == list(range(131, 141)),
        "ten-day sequence drift",
    )
    need(len({m.group(3) for m in marks}) == 10, "lesson titles duplicated")
    for i, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (27 if day <= 135 else 28), f"Day {day} week drift")
        block = lessons[
            mark.end() : marks[i + 1].start() if i + 1 < len(marks) else len(lessons)
        ]
        times = [
            int(v)
            for v in re.findall(
                r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*", block, re.MULTILINE
            )
        ]
        need(
            times == [3, 4, 5, 7, 4, 2] and sum(times) == 25, f"Day {day} timing drift"
        )
        target = re.search(r"^\*\*Target/codes:\*\* ([^\n]+)", block, re.MULTILINE)
        need(
            target
            and set(re.findall(r"\bAC9HSF[KS]\d\d\b", target.group(1)))
            == EXPECTED[day],
            f"Day {day} exact codes drift",
        )
        need(
            "**Child product:**" in block
            and "**Next move:**" in block
            and "**Check/respond · 4 min.**" in block,
            f"Day {day} product/feedback missing",
        )
        need(
            len(set(table_row("LEARNER-CARDS.md", day, 3))) == 3,
            f"Day {day} routes repeated",
        )
        need(
            all("→" in s for s in table_row("PRACTICE-SWAPS.md", day, 2)),
            f"Day {day} swaps not worked",
        )
        table_row("FAMILY-CARER-OPTIONS.md", day, 1)
    need(
        all(
            term in lessons
            for term in (
                "paper museum",
                "caption repair station",
                "caption tent",
                "five-station",
                "COPY",
            )
        ),
        "practical museum progression absent",
    )
    need(
        "not permanent learning-style categories" in read("LEARNER-CARDS.md").lower(),
        "fixed learner-style boundary absent",
    )
    need(
        "no child must disclose" in read("README.md").lower()
        and "skipping has no assessment cost"
        in read("FAMILY-CARER-OPTIONS.md").lower(),
        "privacy/bridge boundary absent",
    )
    need(
        "“Mira said this was a bookmark that everyone used.”" in read("MATERIALS.md"),
        "D lacks explicit attributed misquotation",
    )
    need(
        "I do not know how anyone else used it." in read("MATERIALS.md"),
        "E unknown scope absent",
    )


def checks() -> None:
    """Keep two distinct fresh checks apart from worked keys and practice."""
    learner, key = read("STUDENT-CHECKS.md"), read("teacher/KEY-AND-NEXT.md")
    need(
        re.findall(r"^## Check ([AB]) · Day (\d+)", learner, re.MULTILINE)
        == [("A", "135"), ("B", "140")],
        "fresh schedule drift",
    )
    need(
        learner.count("\n\n1. ") == 2 and "KEY-AND-NEXT" not in learner,
        "public check/key separation drift",
    )
    need(
        "Larch Box worked interpretation" in key
        and "Cove Shelf worked interpretation" in key
        and "public formative checks" in key,
        "teacher key absent",
    )
    ordinary = read("PRACTICE-SWAPS.md") + "\n".join(
        x
        for x in read("LEARNER-CARDS.md").splitlines()
        if not x.startswith(("| 135 |", "| 140 |"))
    )
    for name in ("Tavi", "Ona"):
        need(name not in ordinary, f"fresh speaker leaked to ordinary practice: {name}")
    need(
        "two outlined circles and one dark square" in learner
        and "one triangle and two dark dots" in learner,
        "fresh marks drift",
    )
    need(
        "Does the correction prove that nobody ever wore" in learner
        and "does not prove nobody wore" in key,
        "correction scope prompt/key absent",
    )
    need(
        "not yet observable" in key and "exact assistance" in key,
        "formative support record absent",
    )


def assets() -> None:
    """Check A4 geometry, measured text bounds and every exact alternative."""
    alt = read("print/TEXT-ALTERNATIVES.md")
    tree = ast.parse(read("print/generate_print.py"))
    pages = next(
        node.value
        for node in tree.body
        if isinstance(node, ast.Assign)
        and any(isinstance(t, ast.Name) and t.id == "PAGES" for t in node.targets)
    )
    need(
        isinstance(pages, ast.Dict) and {k.value for k in pages.keys} == set(STEMS),
        "generator page set drift",
    )
    for stem in STEMS:
        top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
        need(
            top.get("width") == "210mm"
            and top.get("height") == "297mm"
            and top.get("viewBox") == "0 0 794 1123"
            and top.find("s:title", NS) is not None
            and top.find("s:desc", NS) is not None
            and top.get("aria-labelledby") == "title desc",
            f"{stem} A4/semantic SVG drift",
        )
        nodes = top.findall(".//s:text", NS)
        exact = compact(" ".join("".join(n.itertext()) for n in nodes))
        section = re.search(
            rf"^## {stem}\.svg / {stem}\.pdf\n\n\*\*Exact visible text:\*\* (.*?)\n\n\*\*Illustration and tactile/no-print:\*\* (.*?)(?=\n\n## |\Z)",
            alt,
            re.MULTILINE | re.DOTALL,
        )
        need(
            section
            and compact(section.group(1)) == exact
            and len(section.group(2).split()) >= 15,
            f"{stem} exact text/tactile alternative drift",
        )
        need(
            len(top.findall(".//s:path", NS))
            + len(top.findall(".//s:circle", NS))
            + len(top.findall(".//s:polygon", NS))
            >= 2,
            f"{stem} illustration absent",
        )
        for n in nodes:
            value = "".join(n.itertext())
            size = int(n.get("font-size", "20"))
            font = ImageFont.truetype(
                str(
                    FONT_ROOT
                    / (
                        "DejaVuSans-Bold.ttf"
                        if n.get("font-weight") == "700"
                        else "DejaVuSans.ttf"
                    )
                ),
                size,
            )
            x, y = float(n.get("x", "0")), float(n.get("y", "0"))
            left, up, right, down = font.getbbox(value, anchor="ls")
            need(
                x + left >= 29
                and x + right <= 765
                and y + up >= 20
                and y + down <= 1100,
                f"{stem} text bounds drift: {value}",
            )
        pdf = ROOT / "print" / f"{stem}.pdf"
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        raw = subprocess.check_output(["pdftotext", "-raw", str(pdf), "-"], text=True)
        fonts = subprocess.check_output(["pdffonts", str(pdf)], text=True)
        need(
            re.search(r"Pages:\s+1\b", info) and "(A4)" in info,
            "PDF A4/page drift " + stem,
        )
        need(
            "DejaVu" in fonts and "yes yes yes" in fonts,
            "embedded/searchable PDF font drift " + stem,
        )
        need(compact(raw) == exact, f"{stem} exact searchable PDF text drift")
    source_material = read("MATERIALS.md")
    check_material = read("STUDENT-CHECKS.md")
    for stem, source in [
        ("paper-marker-archive-card", source_material),
        ("mira-maker-note", source_material),
        ("draft-caption", source_material),
        ("mira-correction", source_material),
        ("check-a-larch-box", check_material),
        ("check-b-cove-shelf", check_material),
    ]:
        top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
        words = compact(
            " ".join("".join(n.itertext()) for n in top.findall(".//s:text", NS))
        )
        for quote in re.findall("“([^”]+)”", words):
            need(
                quote in compact(source.replace("**", "")),
                f"{stem} exact source-script quotation drift: {quote}",
            )
    need(
        "PDFs are not tagged" in alt
        and "bitstream" in read("print/DEJAVU-FONT-LICENSE.txt").lower(),
        "PDF/font claim boundary missing",
    )
    a = ET.parse(ROOT / "print/check-a-larch-box.svg").getroot()
    b = ET.parse(ROOT / "print/check-b-cove-shelf.svg").getroot()
    need(
        len(a.findall(".//s:circle", NS)) == 2
        and len(a.findall(".//s:polygon", NS)) == 1,
        "fresh A marks drift",
    )
    need(
        len(b.findall(".//s:circle", NS)) == 2
        and len(b.findall(".//s:polygon", NS)) == 1,
        "fresh B marks drift",
    )


def links(build: bool) -> int:
    """Validate local Markdown links without network access or mutation."""
    total = 0
    for page in ROOT.rglob("*.md"):
        for link in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", page.read_text()):
            if link.startswith(("https://", "http://", "mailto:")):
                continue
            name, marker, anchor = unquote(link).partition("#")
            target = (page.parent / name).resolve() if name else page
            if not (build and target == ROOT / "manifest.json"):
                need(
                    target.is_file(),
                    f"broken local link {page.relative_to(ROOT)} -> {link}",
                )
            if marker and target.suffix == ".md":
                heads = re.findall(r"^#{1,6} (.+)$", target.read_text(), re.MULTILINE)
                slugs = {
                    re.sub(
                        r"[^\w -]", "", re.sub(r"[*_]", "", h).strip().lower()
                    ).replace(" ", "-")
                    for h in heads
                }
                need(anchor in slugs, f"broken anchor {link}")
            total += 1
    return total


def manifest_data() -> dict[str, object]:
    """Return a complete file receipt; the caller deliberately saves it."""
    files = sorted(
        p
        for p in ROOT.rglob("*")
        if p.is_file()
        and p.name != "manifest.json"
        and "__pycache__" not in p.parts
        and ".ruff_cache" not in p.parts
    )
    names = {p.relative_to(ROOT).as_posix() for p in files}
    need(
        names == REQUIRED,
        f"pack file set drift: missing={sorted(REQUIRED-names)}, extra={sorted(names-REQUIRED)}",
    )
    for file in files:
        if file.suffix in (".md", ".py", ".svg", ".txt", ".json"):
            body = file.read_text()
            need(
                body.endswith("\n")
                and "\r" not in body
                and all(line == line.rstrip() for line in body.splitlines()),
                f"whitespace drift {file.relative_to(ROOT)}",
            )
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/hass/term-3/weeks-27-28",
        "created_at": "2026-09-30",
        "review_status": "author_desk_checked_pending_educator_local_access_and_child_review",
        "curriculum_source_sha256": SHA,
        "curriculum_codes_partial_conditional": sorted(SKILLS),
        "curriculum_codes_fictional_rehearsal_only": [],
        "curriculum_codes_held_for_authentic_local_sources": sorted(HELD),
        "files": {
            p.relative_to(ROOT).as_posix(): {
                "sha256": hashlib.sha256(p.read_bytes()).hexdigest(),
                "bytes": p.stat().st_size,
            }
            for p in files
        },
    }


def main() -> None:
    """Run an offline audit; JSON mode prints a proposed manifest only."""
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument(
        "--manifest-json",
        action="store_true",
        help="Print a proposed receipt; does not write any file",
    )
    args = parser.parse_args()
    sources()
    pedagogy()
    checks()
    assets()
    n = links(args.manifest_json)
    expected = manifest_data()
    if args.manifest_json:
        print(json.dumps(expected, ensure_ascii=False, indent=2))
    else:
        need(
            json.loads(read("manifest.json")) == expected,
            "SHA-256 manifest missing/stale",
        )
        print(
            f"PASS: nine exact direct workbook rows/QCAA pins; ten 25-minute scripts; 30 routes; 20 swaps; 10 bridges; two separate fresh checks; eight A4 pairs/exact alternatives/text bounds; {n} local links; {len(expected['files'])} SHA-256 files"
        )


if __name__ == "__main__":
    main()
