# SPDX-License-Identifier: Apache-2.0
"""Fail-closed, read-only by default receipt for HASS Weeks 15-16."""

from __future__ import annotations

import argparse
import ast
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
ROWS = {
    "AC9HSFK01": 1213,
    "AC9HSFK02": 1217,
    "AC9HSFK03": 1221,
    "AC9HSFK04": 1226,
    "AC9HSFS01": 1233,
    "AC9HSFS02": 1237,
    "AC9HSFS03": 1241,
    "AC9HSFS04": 1245,
    "AC9HSFS05": 1251,
}
HELD_CODES = {"AC9HSFK01", "AC9HSFK02", "AC9HSFK03", "AC9HSFK04"}
SKILL_CODES = set(ROWS) - HELD_CODES
REHEARSAL_CODE = "AC9HSFK02"
QCAA_URL = (
    "https://www.qcaa.qld.edu.au/p-10/aciq/version-9/"
    "learning-areas/p-10-humanities-and-social-sciences/hass"
)
QCAA_ALIGNMENT_URL = (
    "https://www.qcaa.qld.edu.au/downloads/aciqv9/"
    "humanities-and-social-sciences/curriculum/ac9_hass_prep_as_cd_alignment.pdf"
)
STEMS = (
    "story-cart-notice",
    "story-cart-timeline",
    "story-cart-voices",
    "story-cart-objects",
    "event-time-mat",
    "source-voice-mat",
    "check-a-paper-theatre",
    "check-b-message-wall",
)
FINAL_DAY = 80
CHECK_COUNT = 2
MIN_PDF_WORDS = 20
EXPECTED_CODES = {
    71: {"AC9HSFS01", "AC9HSFS04", REHEARSAL_CODE},
    72: {"AC9HSFS02", "AC9HSFS04", REHEARSAL_CODE},
    73: {"AC9HSFS03", "AC9HSFS04", "AC9HSFS05", REHEARSAL_CODE},
    74: {"AC9HSFS01", "AC9HSFS03", "AC9HSFS04"},
    75: set(SKILL_CODES) | {REHEARSAL_CODE},
    76: {"AC9HSFS01", "AC9HSFS02", "AC9HSFS04", REHEARSAL_CODE},
    77: {"AC9HSFS02", "AC9HSFS03", "AC9HSFS04"},
    78: {"AC9HSFS01", "AC9HSFS03", "AC9HSFS04", REHEARSAL_CODE},
    79: set(SKILL_CODES) | {REHEARSAL_CODE},
    80: set(SKILL_CODES) | {REHEARSAL_CODE},
}
PDF_PHRASES = {
    "story-cart-notice": (
        "STORY CART DAY",
        "FIRST YEAR",
        "NEXT YEAR",
        "grandfather Kit",
        "two-part cue",
    ),
    "story-cart-timeline": (
        "FIRST YEAR · OPENING",
        "LATER THAT DAY · READING",
        "NEXT YEAR · MARKING",
        "Cart opens",
        "New covers remember",
    ),
    "story-cart-voices": (
        "MARA",
        "IVO",
        "opening matters to me",
        "only in the next year",
    ),
    "story-cart-objects": (
        "TICKET",
        "FIRST YEAR",
        "DRAWN COVER",
        "NEXT YEAR",
        "object label alone",
    ),
    "event-time-mat": (
        "FIRST OPENING",
        "LATER SAME DAY",
        "NEXT YEAR MARKING",
        "NOT TOLD / QUESTION",
    ),
    "source-voice-mat": (
        "NOTICE A · STATES",
        "PICTURES B · SHOW",
        "MARA C · SAYS",
        "IVO C · SAYS",
        "STILL UNKNOWN",
    ),
    "check-a-paper-theatre": (
        "PAPER THEATRE DAY",
        "FIRST YEAR · STAGE BUILT",
        "LATER THAT DAY · FIRST STORY",
        "NEXT YEAR · NEW TICKETS",
        "ADA",
        "aunt Lou",
    ),
    "check-b-message-wall": (
        "MESSAGE WALL DAY",
        "FIRST YEAR · OPENING",
        "NEXT YEAR · NEW NOTES",
        "SOL",
        "LEE",
        "Some people put up new notes",
    ),
}


def need(ok: object, message: str) -> None:
    """Fail with one actionable explanation."""
    if not ok:
        raise AssertionError(message)


def read(name: str) -> str:
    """Read authored UTF-8 text."""
    return (ROOT / name).read_text(encoding="utf-8")


def source() -> None:
    """Pin exact official rows and Queensland Prep entry points."""
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    imported = json.loads(
        (STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8")
    )
    snapshot = json.loads(read("source-snapshot.json"))
    crosswalk = read("CURRICULUM-CROSSWALK.md")
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == imported["source_sha256"]
        == snapshot["workbook_sha256"]
        == SOURCE_SHA,
        "ACARA official workbook/import/snapshot SHA drift",
    )
    need(snapshot["content_rows"] == ROWS, "ACARA row pin drift")
    need(
        set(snapshot["skill_practice_partial_codes"]) == SKILL_CODES
        and set(snapshot["fictional_concept_rehearsal_only_codes"]) == {REHEARSAL_CODE}
        and set(snapshot["knowledge_codes_held_for_authentic_local_sources"])
        == HELD_CODES,
        "Skill/rehearsal/held source accounting drift",
    )
    need(
        snapshot["queensland_prep_entry_point_url"] == QCAA_URL
        and snapshot["queensland_prep_alignment_url"] == QCAA_ALIGNMENT_URL
        and QCAA_URL in crosswalk
        and QCAA_ALIGNMENT_URL in crosswalk,
        "QCAA Prep URL pin drift",
    )
    records = {
        record["code"]: record
        for record in imported["records"]
        if record.get("record_type") == "content_description"
        and record.get("code") in ROWS
    }
    need(set(records) == set(ROWS), "Foundation HASS rows missing")
    for code, row in ROWS.items():
        item = records[code]
        attrs = item["attributes"]
        need(
            item["source_row"] == row
            and attrs["learning_area"] == "Humanities and Social Sciences"
            and attrs["subject"] == "HASS F-6"
            and attrs["level"] == "Foundation Year",
            f"Official exact record drift: {code}",
        )
        pattern = (
            rf"^\| {code} \| {row} \| Humanities and Social Sciences · HASS F-6 · "
            rf"Foundation Year \| {re.escape(item['plain_text'])} \|"
        )
        need(
            re.search(pattern, crosswalk, re.MULTILINE),
            f"Verbatim crosswalk drift: {code}",
        )
        if code in HELD_CODES:
            row_line = next(
                line
                for line in crosswalk.splitlines()
                if line.startswith(f"| {code} |")
            )
            need(
                "hold" in row_line.lower() or "held" in row_line.lower(),
                f"Authentic local hold absent: {code}",
            )
    need(
        "k02" in crosswalk.lower()
        and "fictional concept rehearsal only" in crosswalk.lower()
        and "not secure exams" in crosswalk.lower()
        and "country/place" in crosswalk.lower(),
        "Crosswalk evidence boundary absent",
    )
    print("PASS nine exact ACARA v9 Foundation HASS rows and QCAA Prep pins")


def sections(body: str, pattern: str) -> list[tuple[int, str]]:
    """Split numbered day sections without accepting duplicates."""
    marks = list(re.finditer(pattern, body, re.MULTILINE))
    return [
        (
            int(mark.group(1)),
            body[
                mark.end() : marks[index + 1].start()
                if index + 1 < len(marks)
                else len(body)
            ],
        )
        for index, mark in enumerate(marks)
    ]


def pedagogy() -> None:
    """Check ten timed scripts, thirty access routes and twenty worked swaps."""
    lessons = read("LESSONS.md")
    teacher_days = sections(lessons, r"^## Week \d+ · Day (\d+)\b")
    need(
        [day for day, _ in teacher_days] == list(range(71, 81)),
        "Teacher day sequence",
    )
    for day, body in teacher_days:
        minutes = [
            int(value)
            for value in re.findall(
                r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*", body, re.MULTILINE
            )
        ]
        need(minutes == [3, 4, 5, 7, 4, 2], f"Day {day} 25-min timing: {minutes}")
        target = re.search(r"^\*\*Target/codes:\*\* ([^\n]+)", body, re.MULTILINE)
        need(target is not None, f"Day {day} target absent")
        codes = set(re.findall(r"\bAC9HSF(?:K|S)\d\d\b", target.group(1)))
        if day == FINAL_DAY:
            need(len(codes) == len(SKILL_CODES) + 1, "Day 80 all sampled codes absent")
        need(codes == EXPECTED_CODES[day], f"Day {day} code drift: {codes}")
        need(
            "**Check/respond" in body
            and any(token in body.lower() for token in ("record", "collect"))
            and "**Exit" in body,
            f"Day {day} feedback/record absent",
        )
    cards = read("LEARNER-CARDS.md")
    learner_days = sections(cards, r"^## Day (\d+)\b")
    need([day for day, _ in learner_days] == list(range(71, 81)), "Learner days")
    for day, body in learner_days:
        need(
            "**Shared target:**" in body
            and all(
                re.search(rf"^- \*\*{route} ·", body, re.MULTILINE) for route in "ABC"
            )
            and "\n\n- **A ·" in body
            and "\n\n**Home:**" in body,
            f"Day {day} routes/home or CommonMark list spacing",
        )
    swaps = read("PRACTICE-SWAPS.md")
    rows = re.findall(r"^\| (\d+) \| (.+) \| (.+) \|$", swaps, re.MULTILINE)
    need(
        [int(day) for day, _, _ in rows] == list(range(71, 81))
        and all(one.startswith("**") and two.startswith("**") for _, one, two in rows),
        "Twenty worked context swaps absent",
    )
    bridges = read("FAMILY-CARER-OPTIONS.md")
    bridge_days = re.findall(
        r"^\| After Day (\d+) \| (.+) \| (.+) \|$", bridges, re.MULTILINE
    )
    need(
        [int(day) for day, _, _ in bridge_days] == list(range(71, 81))
        and all(idea and school for _, idea, school in bridge_days),
        "Ten optional bridge/full school routes absent",
    )
    need(
        "not fixed learning styles" in cards.lower()
        and "no child must" in read("README.md").lower()
        and "cultural authority" in read("LOCAL-SOURCE-INSERT.md").lower(),
        "Access/privacy/cultural boundary absent",
    )
    print("PASS ten 25-minute scripts, 30 routes, 20 worked swaps")


def checks() -> None:
    """Check new public cases, separate key and source-bounded worked answers."""
    student = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    linear_key = " ".join(re.sub(r"[*_]", "", key.lower()).split())
    need(
        re.findall(r"^## Check ([AB]) · Day (\d+)\b", student, re.MULTILINE)
        == [("A", "75"), ("B", "80")],
        "Fresh check schedule drift",
    )
    need(student.count("\n\n1. ") == CHECK_COUNT, "Student list spacing drift")
    need(
        "teacher/KEY-AND-NEXT.md" not in student
        and "teacher/KEY-AND-NEXT.md" not in read("LEARNER-CARDS.md")
        and "publicly accessible" in key.lower()
        and "not secure exams" in student.lower(),
        "Public check/key separation absent",
    )
    routine = read("PRACTICE-SWAPS.md") + " ".join(
        body
        for day, body in sections(read("LEARNER-CARDS.md"), r"^## Day (\d+)\b")
        if day not in (75, 80)
    )
    for leaked in ("Paper Theatre", "Ada", "Aunt Lou", "Message Wall", "Sol", "Lee"):
        need(leaked not in routine, f"Held case leaked into earlier routine: {leaked}")
    for phrase in (
        "first year",
        "later that day",
        "next year",
        "to remember the first puppet story",
        "Ada",
        "Aunt Lou",
        "I was not at the first puppet story",
        "Sol",
        "Lee",
        "did not come back",
        "different participation times",
        "all families",
    ):
        need(phrase.lower() in linear_key, f"Worked key fact missing: {phrase}")
    need(
        "this invented packet records no real family" in student.lower()
        and "no real message wall" in student.lower()
        and "no real child/classroom outcome" in key.lower(),
        "Check fictional/observed boundary absent",
    )
    print("PASS two fresh public formative checks and separate public key")


def assets() -> None:
    """Check eight A4 pairs, distinct drawings and complete linear alternatives."""
    alternatives = read("print/TEXT-ALTERNATIVES.md")
    tree = ast.parse(read("print/generate_print.py"))
    pages = next(
        node.value
        for node in tree.body
        if isinstance(node, ast.Assign)
        and any(
            isinstance(target, ast.Name) and target.id == "PAGES"
            for target in node.targets
        )
    )
    need(
        isinstance(pages, ast.Dict)
        and {key.value for key in pages.keys if isinstance(key, ast.Constant)}
        == set(STEMS),
        "Generated aid inventory drift",
    )
    ns = {"svg": "http://www.w3.org/2000/svg"}
    linear_alt = " ".join(re.sub(r"[*_]", "", alternatives.lower()).split())
    check_pages: dict[str, str] = {}
    for stem in STEMS:
        top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
        pdf = ROOT / "print" / f"{stem}.pdf"
        need(
            top.get("width") == "210mm"
            and top.get("height") == "297mm"
            and top.find("svg:title", ns) is not None
            and top.find("svg:desc", ns) is not None,
            f"A4 SVG/metadata drift: {stem}",
        )
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        content = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        linear_pdf = " ".join(content.lower().split())
        if stem.startswith("check-"):
            check_pages[stem] = linear_pdf
        need(
            re.search(r"Pages:\s+1\b", info)
            and "(A4)" in info
            and "CC BY 4.0" in content
            and len(content.split()) >= MIN_PDF_WORDS
            and f"{stem}.svg" in alternatives
            and f"{stem}.pdf" in alternatives,
            f"PDF/selectable text/full alternative drift: {stem}",
        )
        for phrase in PDF_PHRASES[stem]:
            need(
                phrase.lower() in linear_pdf and phrase.lower() in linear_alt,
                f"PDF/linear source mismatch: {stem}: {phrase}",
            )
    need(
        "every family" not in check_pages["check-a-paper-theatre"]
        and "both attended both years" not in check_pages["check-b-message-wall"]
        and "every family" not in check_pages["check-b-message-wall"],
        "Assessment source page states its own inferred answer",
    )
    timeline = ET.parse(ROOT / "print/story-cart-timeline.svg").getroot()
    notice = ET.parse(ROOT / "print/story-cart-notice.svg").getroot()
    check_b = ET.parse(ROOT / "print/check-b-message-wall.svg").getroot()
    labels = [
        ("".join(node.itertext()), int(node.get("y", "0")))
        for node in timeline.findall(".//svg:text", ns)
    ]
    time_y = {label: y for label, y in labels}
    rectangles_b = [
        (int(node.get("x", "0")), int(node.get("y", "0")))
        for node in check_b.findall(".//svg:rect", ns)
    ]
    need(
        time_y["FIRST YEAR · OPENING"]
        < time_y["LATER THAT DAY · READING"]
        < time_y["NEXT YEAR · MARKING"]
        and len(notice.findall(".//svg:circle", ns)) == 2
        and all(point in rectangles_b for point in ((491, 309), (551, 328), (611, 303)))
        and (95, 309) not in rectangles_b,
        "Fictional time/object geometry drift",
    )
    need(
        "untagged" in alternatives
        and "tactile" in alternatives.lower()
        and "no real family" in alternatives.lower(),
        "Accessible alternative or fictional boundary absent",
    )
    print("PASS eight A4 SVG/PDF pairs, geometry, selectable text and alternatives")


def slug(title: str) -> str:
    """Reproduce local Markdown heading anchors."""
    title = re.sub(r"\[[^]]+\]\([^)]+\)", "", title)
    title = re.sub(r"[*_]", "", title).strip().lower()
    title = re.sub(r"[^\w -]", "", title)
    return title.replace(" ", "-")


def links(*, allow_manifest_bootstrap: bool) -> int:
    """Check local Markdown paths and heading targets."""
    count = 0
    for path in ROOT.rglob("*.md"):
        body = path.read_text(encoding="utf-8")
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", body):
            if url.startswith(("https://", "http://", "mailto:")):
                continue
            name, marker, anchor = unquote(url).partition("#")
            target = (path.parent / name).resolve() if name else path
            if allow_manifest_bootstrap and target == ROOT / "manifest.json":
                count += 1
                continue
            need(target.is_file(), f"Broken link: {path.relative_to(ROOT)} -> {url}")
            if marker and target.suffix.lower() == ".md":
                headings = [
                    slug(value)
                    for value in re.findall(
                        r"^#{1,6} (.+)$",
                        target.read_text(encoding="utf-8"),
                        re.MULTILINE,
                    )
                ]
                need(anchor in headings, f"Broken heading anchor: {url}")
            count += 1
    print(f"PASS {count} local file/heading links")
    return count


def manifest_data() -> dict:
    """Hash authored files, excluding the self-referential manifest."""
    files = sorted(
        path
        for path in ROOT.rglob("*")
        if path.is_file()
        and path.name != "manifest.json"
        and "__pycache__" not in path.parts
        and ".ruff_cache" not in path.parts
    )
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/hass/term-2/weeks-15-16",
        "created_at": "2026-09-30",
        "review_status": "author_desk_checked_pending_school_material_and_child_review",
        "curriculum_source_sha256": SOURCE_SHA,
        "curriculum_codes_partial_conditional": sorted(SKILL_CODES),
        "curriculum_codes_fictional_rehearsal_only": [REHEARSAL_CODE],
        "curriculum_codes_held_for_authentic_local_sources": sorted(HELD_CODES),
        "files": {
            str(path.relative_to(ROOT)): {
                "sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
                "bytes": path.stat().st_size,
            }
            for path in files
        },
    }


def main() -> None:
    """Run offline checks and optionally write a deterministic SHA receipt."""
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    source()
    pedagogy()
    checks()
    assets()
    links(allow_manifest_bootstrap=args.write_manifest)
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
        )
        print(f"WROTE SHA-256 manifest for {len(expected['files'])} files")
    else:
        need(
            json.loads(manifest.read_text(encoding="utf-8")) == expected,
            "SHA-256 manifest missing or stale",
        )
        print(f"PASS SHA-256 manifest for {len(expected['files'])} files")
    print("PACK PASS · author desk QA only; no physical or child trial")


if __name__ == "__main__":
    main()
