# SPDX-License-Identifier: Apache-2.0
"""Fail-closed offline receipt for Foundation HASS Weeks 3-4."""

from __future__ import annotations

import argparse
import ast
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
ROWS = {
    "AC9HSFK01": 1213,
    "AC9HSFK02": 1217,
    "AC9HSFK03": 1221,
    "AC9HSFK04": 1226,
    "AC9HSFS01": 1233,
    "AC9HSFS02": 1237,
    "AC9HSFS03": 1241,
    "AC9HSFS04": 1245,
    "AC9HSFS05": 1251,
}
HELD_CODES = {"AC9HSFK01", "AC9HSFK02", "AC9HSFK04"}
PARTIAL_CODES = set(ROWS) - HELD_CODES
QCAA_URL = (
    "https://www.qcaa.qld.edu.au/p-10/aciq/version-9/"
    "learning-areas/p-10-humanities-and-social-sciences/hass"
)
QCAA_ALIGNMENT_URL = (
    "https://www.qcaa.qld.edu.au/downloads/aciqv9/"
    "humanities-and-social-sciences/curriculum/ac9_hass_prep_as_cd_alignment.pdf"
)
LESSON_MINUTES = 25
SPLIT_DAY = 15
MIN_PDF_WORDS = 20
CHECK_COUNT = 2
STEMS = (
    "paper-harbor-people",
    "relationship-unknown-mat",
    "ren-timeline",
    "paper-harbor-views",
    "paper-lane-maps",
    "lane-views-evidence",
    "check-a-new-people",
    "check-b-paper-nook",
)
EXPECTED_CODES = {
    11: {"AC9HSFS01"},
    12: {"AC9HSFS02", "AC9HSFS05"},
    13: {"AC9HSFS01", "AC9HSFS03"},
    14: {"AC9HSFS02", "AC9HSFS05"},
    15: {"AC9HSFS01", "AC9HSFS02", "AC9HSFS04", "AC9HSFS05"},
    16: {"AC9HSFK03", "AC9HSFS01"},
    17: {"AC9HSFS02", "AC9HSFS05"},
    18: {"AC9HSFS03", "AC9HSFS04"},
    19: {"AC9HSFK03", "AC9HSFS01", "AC9HSFS04"},
    20: {
        "AC9HSFK03",
        "AC9HSFS02",
        "AC9HSFS03",
        "AC9HSFS04",
        "AC9HSFS05",
    },
}
PDF_PHRASES = {
    "paper-harbor-people": (
        "FIO",
        "KIT",
        "REN",
        "VALE",
        "CLOUD TOWN",
        "PEBBLE TOWN",
    ),
    "relationship-unknown-mat": (
        "FAMILY RELATIONSHIP",
        "ROOM ROLE",
        "SOURCE SAYS",
        "SOURCE DOES NOT SAY",
    ),
    "ren-timeline": ("BORN", "CLOUD TOWN", "RAISED", "PEBBLE TOWN"),
    "paper-harbor-views": ("FIO says", "VALE says", "labelled shelf"),
    "paper-lane-maps": (
        "EARLIER",
        "LATER",
        "PAPER TRACK",
        "CATCH BOX",
        "ENTRY",
    ),
    "lane-views-evidence": (
        "FIO says",
        "VALE says",
        "SOURCE DETAIL",
        "WHOSE VIEW?",
        "CONCLUSION / UNKNOWN",
    ),
    "check-a-new-people": (
        "VIVI / OTIS",
        "KIO",
        "REED TOWN",
        "LANTERN TOWN",
        "SANA",
    ),
    "check-b-paper-nook": (
        "EARLIER",
        "LATER",
        "BOOK SPACE",
        "MIKA says",
        "LIO says",
    ),
}


def need(ok: object, message: str) -> None:
    """Fail with an actionable reason."""
    if not ok:
        raise AssertionError(message)


def read(name: str) -> str:
    """Read one authored UTF-8 file."""
    return (ROOT / name).read_text(encoding="utf-8")


def source() -> None:
    """Pin official workbook, imported exact rows, level and QCAA context."""
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    imported = json.loads(
        (STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8")
    )
    snapshot = json.loads(read("source-snapshot.json"))
    crosswalk = read("CURRICULUM-CROSSWALK.md")
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == imported["source_sha256"]
        == snapshot["workbook_sha256"]
        == SOURCE_SHA,
        "Official ACARA workbook/import/snapshot SHA drift",
    )
    need(snapshot["content_rows"] == ROWS, "ACARA row pin drift")
    need(
        set(snapshot["baseline_partial_codes"]) == PARTIAL_CODES
        and set(snapshot["not_claimed_without_local_sources"]) == HELD_CODES,
        "Partial/held source accounting drift",
    )
    need(
        snapshot["queensland_prep_entry_point_url"] == QCAA_URL
        and snapshot["queensland_prep_alignment_url"] == QCAA_ALIGNMENT_URL
        and QCAA_URL in crosswalk
        and QCAA_ALIGNMENT_URL in crosswalk,
        "QCAA source pin drift",
    )
    records = {
        record["code"]: record
        for record in imported["records"]
        if record.get("record_type") == "content_description"
        and record.get("code") in ROWS
    }
    need(set(records) == set(ROWS), "Foundation HASS rows missing")
    for code, row in ROWS.items():
        item = records[code]
        attrs = item["attributes"]
        need(
            item["source_row"] == row
            and attrs["learning_area"] == "Humanities and Social Sciences"
            and attrs["subject"] == "HASS F-6"
            and attrs["level"] == "Foundation Year",
            f"Official exact record drift: {code}",
        )
        pattern = (
            rf"^\| {code} \| {row} \| Humanities and Social Sciences · HASS F-6 · "
            rf"Foundation Year \| {re.escape(item['plain_text'])} \|"
        )
        need(
            re.search(pattern, crosswalk, re.MULTILINE),
            f"Verbatim crosswalk drift: {code}",
        )
        if code in HELD_CODES:
            need(
                re.search(rf"^\| {code} \|.*\*\*Hold:", crosswalk, re.MULTILINE),
                f"Local/privacy/cultural hold absent: {code}",
            )
    need(
        "k01, k02 and k04 remain open" in crosswalk.lower()
        and "country/place" in crosswalk.lower()
        and "not secure exams" in crosswalk.lower(),
        "Source/evidence boundaries absent",
    )
    print("PASS nine exact ACARA v9 Foundation HASS rows and QCAA Prep pins")


def pedagogy() -> None:
    """Check ten distinct timed scripts, thirty routes and twenty worked swaps."""
    lessons = read("LESSONS.md")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", lessons, re.MULTILINE))
    need(
        [int(mark.group(2)) for mark in marks] == list(range(11, 21)),
        "Teacher Day 11-20 sequence drift",
    )
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        end = marks[index + 1].start() if index + 1 < len(marks) else len(lessons)
        body = lessons[mark.end() : end]
        minutes = [
            int(value)
            for value in re.findall(
                r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*", body, re.MULTILINE
            )
        ]
        need(
            int(mark.group(1)) == (3 if day <= SPLIT_DAY else 4)
            and minutes == [3, 4, 5, 7, 4, 2]
            and sum(minutes) == LESSON_MINUTES,
            f"Day {day} week/timing drift: {minutes}",
        )
        target = re.search(r"^\*\*Target/codes:\*\* ([^\n]+)", body, re.MULTILINE)
        need(target is not None, f"Day {day} target absent")
        codes = set(re.findall(r"\bAC9HSF(?:K|S)\d\d\b", target.group(1)))
        need(codes == EXPECTED_CODES[day], f"Day {day} target code drift: {codes}")
        need(
            "**Check/respond" in body
            and any(word in body.lower() for word in ("record", "capture", "collect")),
            f"Day {day} feedback/record action absent",
        )
    cards = read("LEARNER-CARDS.md")
    card_marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need(
        [int(mark.group(1)) for mark in card_marks] == list(range(11, 21)),
        "Learner Day 11-20 sequence drift",
    )
    for index, mark in enumerate(card_marks):
        end = (
            card_marks[index + 1].start() if index + 1 < len(card_marks) else len(cards)
        )
        body = cards[mark.end() : end]
        need(
            all(re.search(rf"^- \*\*{route} ·", body, re.MULTILINE) for route in "ABC")
            and "**Shared target:**" in body
            and "\n\n- **A ·" in body
            and "\n\n**Home:**" in body
            and "**Home:**" in body,
            f"Day {mark.group(1)} three routes/home absent",
        )
    swaps = read("PRACTICE-SWAPS.md")
    rows = re.findall(r"^\| (\d+) \| (.+) \| (.+) \|$", swaps, re.MULTILINE)
    need(
        [int(day) for day, _, _ in rows] == list(range(11, 21))
        and all(one.startswith("**") and two.startswith("**") for _, one, two in rows),
        "Twenty worked context swaps absent",
    )
    need(
        "not fixed learning styles" in cards
        and "no child must reveal" in read("README.md").lower()
        and "K01/K02/K04" in lessons
        and "cultural authority" in read("LOCAL-SOURCE-INSERT.md").lower(),
        "Access/privacy/cultural boundary absent",
    )
    print("PASS ten 25-minute scripts, 30 routes and 20 worked swaps")


def checks() -> None:
    """Protect first-response freshness and public-key honesty."""
    student = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    routine = read("LEARNER-CARDS.md") + read("PRACTICE-SWAPS.md")
    need(
        re.findall(r"^## Check ([AB]) · Day (\d+)\b", student, re.MULTILINE)
        == [("A", "15"), ("B", "20")],
        "Fresh check schedule drift",
    )
    need(
        student.count("\n\n1. ") == CHECK_COUNT,
        "Child check ordered-list spacing drift",
    )
    need(
        "teacher/KEY-AND-NEXT.md" not in student
        and "teacher/KEY-AND-NEXT.md" not in read("LEARNER-CARDS.md")
        and "public by url" in key.lower()
        and "not secure exams" in student.lower(),
        "Public check/key separation absent",
    )
    for leaked in (
        "Vivi and Otis",
        "Kio",
        "Sana",
        "Mika",
        "Lio",
        "Paper Nook",
    ):
        need(leaked not in routine, f"Held source leaked into routine: {leaked}")
    for phrase in (
        "Vivi and Otis",
        "cousins",
        "Reed Town",
        "Lantern Town",
        "Sana",
        "BOX moved from centre earlier to left later",
        "Mika",
        "Lio",
        "not secure exams",
    ):
        need(phrase.lower() in key.lower(), f"Worked key/source drift: {phrase}")
    need(
        "adult" in key.lower()
        and "not an observed real visitor response" in key.lower()
        and "No actual family" in student,
        "Privacy/source limit absent",
    )
    print("PASS two fresh public checks and separate source-limited public key")


def svg_order(stem: str, attribute: str) -> list[str]:
    """Read source region metadata in SVG document order."""
    top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
    ns = {"svg": "http://www.w3.org/2000/svg"}
    return [
        node.get(attribute, "")
        for node in top.findall(".//svg:rect", ns)
        if node.get(attribute) is not None
    ]


def assets() -> None:
    """Verify exact diagram data, A4 PDF text and linear equivalents."""
    alternatives = read("print/TEXT-ALTERNATIVES.md")
    tree = ast.parse(read("print/generate_print.py"))
    pages = next(
        node.value
        for node in tree.body
        if isinstance(node, ast.Assign)
        and any(
            isinstance(target, ast.Name) and target.id == "PAGES"
            for target in node.targets
        )
    )
    need(
        isinstance(pages, ast.Dict)
        and {key.value for key in pages.keys if isinstance(key, ast.Constant)}
        == set(STEMS),
        "Generated asset inventory drift",
    )
    ns = {"svg": "http://www.w3.org/2000/svg"}
    linear_alt = " ".join(re.sub(r"[*_]", "", alternatives.lower()).split())
    for stem in STEMS:
        top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
        pdf = ROOT / "print" / f"{stem}.pdf"
        need(
            top.get("width") == "210mm"
            and top.get("height") == "297mm"
            and top.find("svg:title", ns) is not None
            and top.find("svg:desc", ns) is not None,
            f"A4 SVG/metadata drift: {stem}",
        )
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        content = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        linear_pdf = " ".join(content.lower().split())
        need(
            re.search(r"Pages:\s+1\b", info)
            and "(A4)" in info
            and "CC BY 4.0" in content
            and len(content.split()) >= MIN_PDF_WORDS
            and f"{stem}.svg" in alternatives
            and f"{stem}.pdf" in alternatives,
            f"PDF/selectable text/full alternative drift: {stem}",
        )
        for phrase in PDF_PHRASES[stem]:
            need(
                phrase.lower() in linear_pdf and phrase.lower() in linear_alt,
                f"PDF/linear source mismatch: {stem}: {phrase}",
            )
    need(
        svg_order("paper-harbor-people", "data-person") == ["FIO", "KIT", "REN", "VALE"]
        and svg_order("ren-timeline", "data-frame")
        == ["BORN CLOUD TOWN", "RAISED PEBBLE TOWN"]
        and svg_order("paper-harbor-views", "data-speaker") == ["FIO", "VALE"]
        and svg_order("paper-lane-maps", "data-stage") == ["EARLIER", "LATER"]
        and svg_order("paper-lane-maps", "data-place")
        == [
            "PAPER TRACK CENTRE",
            "CATCH BOX RIGHT",
            "ENTRY BOTTOM",
            "PAPER TRACK LEFT",
            "CATCH BOX RIGHT",
            "ENTRY BOTTOM",
        ]
        and svg_order("lane-views-evidence", "data-speaker") == ["FIO", "VALE"]
        and svg_order("check-a-new-people", "data-fact")
        == [
            "VIVI OTIS COUSINS",
            "KIO OTIS CARER",
            "KIO BORN REED RAISED LANTERN",
            "SANA BOOK DESK HELPER",
        ]
        and svg_order("check-b-paper-nook", "data-stage") == ["EARLIER", "LATER"]
        and svg_order("check-b-paper-nook", "data-place")
        == [
            "BOX CENTRE",
            "BOOK SPACE RIGHT",
            "ENTRY BOTTOM",
            "BOX LEFT",
            "BOOK SPACE RIGHT",
            "ENTRY BOTTOM",
        ]
        and svg_order("check-b-paper-nook", "data-speaker") == ["MIKA", "LIO"],
        "Original source order/geometry metadata drift",
    )
    need(
        "untagged" in alternatives
        and "tactile" in alternatives.lower()
        and "no gate/result" in alternatives.lower(),
        "Accessibility or result boundary absent",
    )
    print("PASS eight A4 SVG/PDF pairs with source order and text/tactile routes")


def slug(title: str) -> str:
    """Reproduce simple Markdown heading anchors."""
    title = re.sub(r"\[[^]]+\]\([^)]+\)", "", title)
    title = re.sub(r"[*_]", "", title).strip().lower()
    title = re.sub(r"[^\w -]", "", title)
    return title.replace(" ", "-")


def links(*, allow_manifest_bootstrap: bool) -> int:
    """Check all relative Markdown links and local heading targets."""
    count = 0
    for path in ROOT.rglob("*.md"):
        body = path.read_text(encoding="utf-8")
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", body):
            if url.startswith(("https://", "http://", "mailto:")):
                continue
            name, marker, anchor = unquote(url).partition("#")
            target = (path.parent / name).resolve() if name else path
            if allow_manifest_bootstrap and target == ROOT / "manifest.json":
                count += 1
                continue
            need(target.is_file(), f"Broken link: {path.relative_to(ROOT)} -> {url}")
            if marker and target.suffix.lower() == ".md":
                headings = [
                    slug(value)
                    for value in re.findall(
                        r"^#{1,6} (.+)$",
                        target.read_text(encoding="utf-8"),
                        re.MULTILINE,
                    )
                ]
                need(anchor in headings, f"Broken heading anchor: {url}")
            count += 1
    print(f"PASS {count} local file/heading links")
    return count


def manifest_data() -> dict:
    """Hash all authored files, excluding self-referential manifest."""
    files = sorted(
        path
        for path in ROOT.rglob("*")
        if path.is_file()
        and path.name != "manifest.json"
        and "__pycache__" not in path.parts
    )
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/hass/term-1/weeks-03-04",
        "created_at": "2026-09-29",
        "review_status": "author_desk_checked_pending_school_material_and_child_review",
        "curriculum_source_sha256": SOURCE_SHA,
        "curriculum_codes_partial_conditional": sorted(ROWS),
        "files": {
            str(path.relative_to(ROOT)): {
                "sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
                "bytes": path.stat().st_size,
            }
            for path in files
        },
    }


def main() -> None:
    """Run offline validation and optionally freeze its SHA receipt."""
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    source()
    pedagogy()
    checks()
    assets()
    links(allow_manifest_bootstrap=args.write_manifest)
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
        )
        print(f"WROTE SHA-256 manifest for {len(expected['files'])} files")
    else:
        need(
            json.loads(manifest.read_text(encoding="utf-8")) == expected,
            "SHA-256 manifest missing or stale",
        )
        print(f"PASS SHA-256 manifest for {len(expected['files'])} files")
    print("PACK PASS · author desk QA only; no physical or child trial")


if __name__ == "__main__":
    main()
