# SPDX-License-Identifier: Apache-2.0
"""Fail-closed offline receipt for Foundation HASS Term 1 Weeks 1-2."""

from __future__ import annotations

import argparse
import ast
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from collections import Counter
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
QCAA_URL = (
    "https://www.qcaa.qld.edu.au/p-10/aciq/version-9/"
    "learning-areas/p-10-humanities-and-social-sciences/hass"
)
QCAA_ALIGNMENT_URL = (
    "https://www.qcaa.qld.edu.au/downloads/aciqv9/"
    "humanities-and-social-sciences/curriculum/ac9_hass_prep_as_cd_alignment.pdf"
)
ROWS = {
    "AC9HSFK01": 1213,
    "AC9HSFK02": 1217,
    "AC9HSFK03": 1221,
    "AC9HSFK04": 1226,
    "AC9HSFS01": 1233,
    "AC9HSFS02": 1237,
    "AC9HSFS03": 1241,
    "AC9HSFS04": 1245,
    "AC9HSFS05": 1251,
}
HELD_CODES = {"AC9HSFK01", "AC9HSFK02", "AC9HSFK04"}
TAUGHT_CODES = set(ROWS) - HELD_CODES
WEEK_SPLIT_DAY = 5
LESSON_MINUTES = 25
MIN_PDF_WORDS = 18
STEMS = (
    "paper-learning-place-map",
    "fictional-views-and-care-cards",
    "ask-map-source-mat",
    "picture-record-mat",
    "source-and-conclusion-mat",
    "original-sign-bank",
    "sign-making-timeline",
    "make-a-sign-mat",
    "check-a-fresh-place",
    "check-b-fresh-signs",
)
EXPECTED_CODES = {
    1: {"AC9HSFK03", "AC9HSFS01"},
    2: {"AC9HSFK03", "AC9HSFS02"},
    3: {"AC9HSFK03", "AC9HSFS03"},
    4: {"AC9HSFK03", "AC9HSFS04", "AC9HSFS05"},
    5: {"AC9HSFK03", "AC9HSFS02", "AC9HSFS03", "AC9HSFS04", "AC9HSFS05"},
    6: {"AC9HSFK03", "AC9HSFS01", "AC9HSFS05"},
    7: {"AC9HSFS02", "AC9HSFS05"},
    8: {"AC9HSFS03", "AC9HSFS04", "AC9HSFS05"},
    9: {"AC9HSFK03", "AC9HSFS01", "AC9HSFS05"},
    10: {
        "AC9HSFK03",
        "AC9HSFS01",
        "AC9HSFS02",
        "AC9HSFS03",
        "AC9HSFS04",
        "AC9HSFS05",
    },
}
MAIN_MAP = {
    ("ENTRY", "1", "1"),
    ("STORY MAT", "1", "3"),
    ("MAKER TABLE", "2", "2"),
    ("REST SEAT", "3", "1"),
    ("WATER POINT", "3", "3"),
}
CHECK_A_MAP = {
    ("ENTRY", "1", "1"),
    ("CARD RACK", "1", "2"),
    ("NEST SEAT", "1", "3"),
    ("MODEL TABLE", "2", "1"),
    ("WINDOW PAD", "2", "3"),
}
CHECK_B_MAP = {
    ("ENTRY", "1", "1"),
    ("DRAW WALL", "1", "2"),
    ("CARD DESK", "2", "1"),
    ("REST COVE", "2", "2"),
}
PRINTED_SOURCE_PHRASES = {
    "fictional-views-and-care-cards": (
        "The REST SEAT is special to me in this paper place because I can sit "
        "and look at a book quietly.",
        "The MAKER TABLE is special to me in this paper place because I can try "
        "a new paper design there.",
        "A paper direction sign lies across a drawn walkway. An adult checks "
        "the walkway and puts the sign back where it is visible beside STORY MAT.",
    ),
    "original-sign-bank": ("I am looking for the STORY MAT on this paper map.",),
    "check-a-fresh-place": (
        "NEST SEAT matters: I can look at picture cards quietly.",
        "MODEL TABLE matters: I can build paper shapes.",
        "A paper name label slipped away from CARD RACK. An adult put it back "
        "there so it could be read.",
    ),
    "check-b-fresh-signs": ("I am looking for CARD DESK to get a blank paper card.",),
}


def need(ok: object, message: str) -> None:
    """Fail with a useful artifact location or reason."""
    if not ok:
        raise AssertionError(message)


def read(name: str) -> str:
    """Read one authored UTF-8 file."""
    return (ROOT / name).read_text(encoding="utf-8")


def source() -> None:
    """Pin workbook bytes, all nine verbatim rows and two QCAA entry links."""
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    data = json.loads(
        (STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8")
    )
    snapshot = json.loads(read("source-snapshot.json"))
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest() == SOURCE_SHA,
        "Official ACARA workbook SHA drift",
    )
    need(
        data["source_sha256"] == snapshot["workbook_sha256"] == SOURCE_SHA,
        "ACARA import/workbook hash disagreement",
    )
    need(snapshot["content_rows"] == ROWS, "Source row pin drift")
    need(
        snapshot["queensland_prep_entry_point_url"] == QCAA_URL
        and snapshot["queensland_prep_alignment_url"] == QCAA_ALIGNMENT_URL,
        "QCAA Prep link pin drift",
    )
    need(
        set(snapshot["baseline_partial_codes"]) == TAUGHT_CODES
        and set(snapshot["not_claimed_without_local_sources"]) == HELD_CODES,
        "Baseline/hold code accounting drift",
    )
    crosswalk = read("CURRICULUM-CROSSWALK.md")
    need(
        QCAA_URL in crosswalk
        and QCAA_ALIGNMENT_URL in crosswalk
        and "partial" in crosswalk.lower(),
        "Queensland/partial claim boundary absent",
    )
    records = {
        record["code"]: record
        for record in data["records"]
        if record.get("record_type") == "content_description"
        and record.get("code") in ROWS
    }
    need(set(records) == set(ROWS), "Missing official Foundation HASS row")
    for code, row_number in ROWS.items():
        record = records[code]
        attrs = record["attributes"]
        need(
            record["source_row"] == row_number
            and attrs["level"] == "Foundation Year"
            and attrs["learning_area"] == "Humanities and Social Sciences"
            and attrs["subject"] == "HASS F-6",
            f"Wrong HASS level/subject/row: {code}",
        )
        row_pattern = (
            rf"^\| {code} \| {row_number} \| "
            rf"Humanities and Social Sciences · HASS F-6 · Foundation Year \| "
            rf"{re.escape(record['plain_text'])} \|"
        )
        need(
            re.search(row_pattern, crosswalk, re.MULTILINE),
            f"Crosswalk lacks verbatim official wording: {code}",
        )
        if code in HELD_CODES:
            need(
                re.search(
                    rf"^\| {code} \|.*\*\*Hold: not taught or assessed",
                    crosswalk,
                    re.MULTILINE,
                ),
                f"Local/privacy hold absent: {code}",
            )
    print("PASS nine exact official ACARA v9 Foundation HASS rows and QCAA Prep links")


def pedagogy() -> None:
    """Check ten scripts, 30 equivalent routes and 20 worked context swaps."""
    lessons = read("LESSONS.md")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", lessons, re.MULTILINE))
    need(
        [int(mark.group(2)) for mark in marks] == list(range(1, 11)),
        "Teacher day sequence drift",
    )
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        end = marks[index + 1].start() if index + 1 < len(marks) else len(lessons)
        body = lessons[mark.end() : end]
        times = [
            int(value)
            for value in re.findall(
                r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*",
                body,
                re.MULTILINE,
            )
        ]
        need(
            int(mark.group(1)) == (1 if day <= WEEK_SPLIT_DAY else 2)
            and times == [3, 4, 5, 7, 4, 2]
            and sum(times) == LESSON_MINUTES,
            f"Day {day} week/minutes drift: {times}",
        )
        codes = set(re.findall(r"\bAC9HSF(?:K|S)\d\d\b", body))
        need(codes == EXPECTED_CODES[day], f"Day {day} code drift: {codes}")
        need("check" in body.lower(), f"Day {day} evidence check missing")
    cards = read("LEARNER-CARDS.md")
    card_marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need(
        [int(mark.group(1)) for mark in card_marks] == list(range(1, 11)),
        "Clean learner day sequence drift",
    )
    for index, mark in enumerate(card_marks):
        end = (
            card_marks[index + 1].start() if index + 1 < len(card_marks) else len(cards)
        )
        body = cards[mark.end() : end]
        need(
            all(re.search(rf"^\- \*\*{route} ·", body, re.MULTILINE) for route in "ABC")
            and "**Same target:**" in body
            and "**Home:**" in body,
            f"Day {mark.group(1)} route/shared target/home absent",
        )
    swaps = read("PRACTICE-SWAPS.md")
    for day in range(1, 11):
        need(
            re.search(rf"^\| {day} \| \*\*[^|]+ \| \*\*[^|]+ \|", swaps, re.MULTILINE),
            f"Two worked context swaps absent on Day {day}",
        )
    need(
        "not fixed learning styles" in cards
        and "No real home" in read("STUDENT-CHECKS.md")
        and "K01/K02/K04" in lessons,
        "Access/privacy/cultural hold drift",
    )
    print("PASS ten distinct 25-minute scripts, 30 routes and 20 worked swaps")


def checks() -> None:
    """Check source/key agreement, freshness and public assessment boundary."""
    student = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    practice = read("PRACTICE-SWAPS.md") + read("LEARNER-CARDS.md")
    need(
        re.findall(r"^## Check ([AB]) · Day (\d+)\b", student, re.MULTILINE)
        == [("A", "5"), ("B", "10")],
        "Fresh formative check order drift",
    )
    need(
        "teacher/KEY-AND-NEXT.md" not in student
        and "teacher/KEY-AND-NEXT.md" not in read("LEARNER-CARDS.md"),
        "Learner-facing page links to public teacher key",
    )
    need(
        "not secure exams" in student
        and "public by URL" in key
        and "first child-controlled" in student,
        "Public check/first response boundary absent",
    )
    need(
        "three sign pictures are mixed up" in student
        and "order blank sign" not in student,
        "Check B sequence disclosed in child instructions",
    )
    for phrase in (
        "CARD RACK is top-middle",
        "NEST SEAT/quiet picture-card looking to Lumi",
        "MODEL TABLE/building paper shapes to Oko",
        "An adult put it back there so it could be read.",
        "CARD DESK is bottom-left",
        (
            "blank sign → rectangle-card icon and CARD DESK words added "
            "→ sign placed beside CARD DESK"
        ),
        "actual visitor feedback",
        "effect is **unknown**",
    ):
        need(phrase in key, f"Private-key source/limit phrase drift: {phrase}")
    for phrase in (
        "Paper Kite Room",
        "Paper Workshop Pavilion",
        "NEST SEAT",
        "CARD DESK",
    ):
        need(
            phrase not in practice,
            f"Held check source leaked into routine practice: {phrase}",
        )
    need(
        all(
            phrase in read("print/TEXT-ALTERNATIVES.md")
            for phrase in (
                "CARD RACK",
                "NEST SEAT",
                "MODEL TABLE",
                "CARD DESK",
                "DRAW WALL",
                "REST COVE",
            )
        ),
        "Held print/text source disagreement",
    )
    print("PASS two new public formative checks, separate public key and source limits")


def svg_places(stem: str) -> set[tuple[str, str, str]]:
    """Read exact labelled cells from a rendered SVG source page."""
    top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
    namespace = {"svg": "http://www.w3.org/2000/svg"}
    return {
        (item.get("data-place"), item.get("data-row"), item.get("data-col"))
        for item in top.findall(".//svg:rect", namespace)
        if item.get("data-place") is not None
    }


def svg_rect_values(stem: str, attribute: str) -> Counter:
    """Read machine-visible source-card/timeline attributes."""
    top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
    namespace = {"svg": "http://www.w3.org/2000/svg"}
    return Counter(
        item.get(attribute)
        for item in top.findall(".//svg:rect", namespace)
        if item.get(attribute) is not None
    )


def assets() -> None:
    """Check ten physical A4 pairs, selectable text and exact source maps."""
    alternatives = read("print/TEXT-ALTERNATIVES.md")
    tree = ast.parse(read("print/generate_print.py"))
    page_map = next(
        node.value
        for node in tree.body
        if isinstance(node, ast.Assign)
        and any(
            isinstance(target, ast.Name) and target.id == "PAGES"
            for target in node.targets
        )
    )
    need(
        isinstance(page_map, ast.Dict)
        and {key.value for key in page_map.keys if isinstance(key, ast.Constant)}
        == set(STEMS),
        "Print generator page inventory drift",
    )
    namespace = {"svg": "http://www.w3.org/2000/svg"}
    for stem in STEMS:
        top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
        pdf = ROOT / "print" / f"{stem}.pdf"
        need(
            top.get("width") == "210mm"
            and top.get("height") == "297mm"
            and top.find("svg:title", namespace) is not None
            and top.find("svg:desc", namespace) is not None,
            f"SVG A4 or accessible metadata drift: {stem}",
        )
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        body = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        linear_pdf = " ".join(body.split())
        need(
            re.search(r"Pages:\s+1\b", info)
            and "(A4)" in info
            and "CC BY 4.0" in body
            and len(body.split()) >= MIN_PDF_WORDS,
            f"PDF page/selectable-text drift: {stem}",
        )
        need(
            f"{stem}.svg" in alternatives and f"{stem}.pdf" in alternatives,
            f"Full linear/tactile alternative missing: {stem}",
        )
        for phrase in PRINTED_SOURCE_PHRASES.get(stem, ()):
            need(
                phrase in linear_pdf and phrase in alternatives,
                f"Printed source/text alternative drift: {stem}: {phrase}",
            )
    need(svg_places("paper-learning-place-map") == MAIN_MAP, "Main map cell drift")
    need(svg_places("check-a-fresh-place") == CHECK_A_MAP, "Check A map cell drift")
    need(svg_places("check-b-fresh-signs") == CHECK_B_MAP, "Check B map cell drift")
    need(
        svg_rect_values("fictional-views-and-care-cards", "data-source-card")
        == Counter({"PIP": 1, "REN": 1, "CARE": 1}),
        "Main speaker/care card drift",
    )
    need(
        svg_rect_values("check-a-fresh-place", "data-source-card")
        == Counter({"LUMI": 1, "OKO": 1, "CARE": 1}),
        "Check A speaker/care card drift",
    )
    need(
        svg_rect_values("check-b-fresh-signs", "data-source-card")
        == Counter({"VISITOR": 1}),
        "Check B visitor source drift",
    )
    for stem in ("sign-making-timeline", "check-b-fresh-signs"):
        need(
            svg_rect_values(stem, "data-step") == Counter({"1": 1, "2": 1, "3": 1}),
            f"Three source timeline frames drift: {stem}",
        )
    check_b_top = ET.parse(ROOT / "print/check-b-fresh-signs.svg").getroot()
    printed_order = [
        item.get("data-step")
        for item in check_b_top.findall(".//svg:rect", namespace)
        if item.get("data-step") is not None
    ]
    need(printed_order == ["3", "1", "2"], "Check B timeline is no longer mixed")
    need(
        "untagged" in alternatives and "tactile" in alternatives.lower(),
        "PDF/tactile access boundary absent",
    )
    print("PASS ten original A4 SVG/PDF pairs and exact map/source geometry")


def slug(value: str) -> str:
    """Resolve a simple local Markdown heading anchor."""
    value = re.sub(r"\[[^]]+\]\([^)]+\)", "", value)
    value = re.sub(r"[*_]", "", value).strip().lower()
    value = re.sub(r"[^\w -]", "", value)
    return value.replace(" ", "-")


def links(*, allow_manifest_bootstrap: bool = False) -> int:
    """Check local file paths and heading anchors without network requests."""
    count = 0
    for path in ROOT.rglob("*.md"):
        body = path.read_text(encoding="utf-8")
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", body):
            if url.startswith(("https://", "http://", "mailto:")):
                continue
            name, marker, anchor = unquote(url).partition("#")
            target = (path.parent / name).resolve() if name else path
            if allow_manifest_bootstrap and target == ROOT / "manifest.json":
                count += 1
                continue
            need(
                target.is_file(),
                f"Broken local link {path.relative_to(ROOT)} -> {url}",
            )
            if marker and target.suffix.lower() == ".md":
                headings = [
                    slug(title)
                    for title in re.findall(
                        r"^#{1,6} (.+)$",
                        target.read_text(encoding="utf-8"),
                        re.MULTILINE,
                    )
                ]
                need(anchor in headings, f"Broken heading anchor: {url}")
            count += 1
    print(f"PASS {count} local file and heading links")
    return count


def manifest_data() -> dict:
    """Hash every authored file except the self-referential manifest."""
    files = sorted(
        path
        for path in ROOT.rglob("*")
        if path.is_file()
        and path.name != "manifest.json"
        and "__pycache__" not in path.parts
    )
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/hass/term-1/weeks-01-02",
        "created_at": "2026-09-29",
        "review_status": (
            "author_desk_checked_pending_teacher_child_"
            "cultural_authority_accessibility_local_review"
        ),
        "curriculum_source_sha256": SOURCE_SHA,
        "curriculum_codes_partial": sorted(TAUGHT_CODES),
        "curriculum_codes_held": sorted(HELD_CODES),
        "files": {
            str(path.relative_to(ROOT)): {
                "sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
                "bytes": path.stat().st_size,
            }
            for path in files
        },
    }


def main() -> None:
    """Run offline quality gates and optionally freeze exact file hashes."""
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    source()
    pedagogy()
    checks()
    assets()
    links(allow_manifest_bootstrap=args.write_manifest)
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n",
            encoding="utf-8",
        )
        print(f"WROTE SHA-256 manifest for {len(expected['files'])} files")
    else:
        need(
            json.loads(manifest.read_text(encoding="utf-8")) == expected,
            "SHA-256 manifest missing or stale",
        )
        print(f"PASS SHA-256 manifest for {len(expected['files'])} files")
    print("PACK PASS · authored offline artifacts; no live classroom/access validation")


if __name__ == "__main__":
    main()
