# SPDX-License-Identifier: Apache-2.0
"""Fail-closed offline receipt for Foundation Digital Technologies Weeks 1-2."""

from __future__ import annotations

import argparse
import ast
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from collections import Counter
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
WEEK_SPLIT_DAY = 5
LESSON_MINUTES = 25
MIN_PDF_WORDS = 18
VIEW_BUTTONS = 3
ROWS = {"AC9TDIFK01": 20098, "AC9TDIFK02": 20105, "AC9TDIFP01": 20111}
QCAA_URL = (
    "https://www.qcaa.qld.edu.au/p-10/aciq/version-9/"
    "learning-areas/p-10-technologies/digital-technologies"
)
QCAA_ALIGNMENT_URL = (
    "https://www.qcaa.qld.edu.au/downloads/aciqv9/technologies/"
    "curriculum/ac9_tech_digital_prep_as_cd_alignment.pdf"
)
STEMS = (
    "fictional-shelf-source",
    "object-picture-symbol-mat",
    "hardware-software-purpose",
    "personal-data-choice-cards",
    "original-picture-symbol-bank",
    "source-representation-system-mat",
    "check-a-fresh-rack",
    "check-b-fresh-props",
)
EXPECTED_CODES = {
    1: {"AC9TDIFK02"},
    2: {"AC9TDIFK02"},
    3: {"AC9TDIFK01", "AC9TDIFK02"},
    4: {"AC9TDIFK02", "AC9TDIFP01"},
    5: {"AC9TDIFK02", "AC9TDIFP01"},
    6: {"AC9TDIFK02"},
    7: {"AC9TDIFK01", "AC9TDIFK02"},
    8: {"AC9TDIFP01"},
    9: {"AC9TDIFK02", "AC9TDIFP01"},
    10: {"AC9TDIFK02", "AC9TDIFP01"},
}
PRINTED_PHRASES = {
    "fictional-shelf-source": ("BOOK, BOOK, CUP", "Two BOOK. One CUP."),
    "hardware-software-purpose": (
        "HARDWARE EXAMPLE · SCREEN",
        "HARDWARE EXAMPLE · KEYBOARD",
        "SOFTWARE EXAMPLE · OFFLINE APP",
    ),
    "personal-data-choice-cards": (
        "MY NAME",
        "MY FACE PHOTO",
        "MY VOICE RECORDING",
        "PLAIN SQUARE ICON",
    ),
    "original-picture-symbol-bank": ("BOOK", "CUP", "MAT"),
    "check-a-fresh-rack": (
        "HAT, HAT, BAG",
        "MY NAME",
        "PLAIN CIRCLE ICON",
    ),
    "check-b-fresh-props": (
        "FLAG, ROUND TOKEN, ROUND TOKEN",
        "F F R",
        "MY FACE PHOTO",
        "PLAIN TRIANGLE ICON",
    ),
}


def need(ok: object, message: str) -> None:
    """Stop on any content, link or hash drift."""
    if not ok:
        raise AssertionError(message)


def read(name: str) -> str:
    """Read UTF-8 authored text."""
    return (ROOT / name).read_text(encoding="utf-8")


def source() -> None:
    """Pin the real workbook and exact Foundation Digital Technologies rows."""
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    imported = json.loads(
        (STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8")
    )
    snapshot = json.loads(read("source-snapshot.json"))
    crosswalk = read("CURRICULUM-CROSSWALK.md")
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == imported["source_sha256"]
        == snapshot["workbook_sha256"]
        == SOURCE_SHA,
        "Official workbook/import/snapshot SHA drift",
    )
    need(snapshot["content_rows"] == ROWS, "Exact official row pin drift")
    need(
        snapshot["queensland_prep_entry_point_url"] == QCAA_URL
        and snapshot["queensland_prep_alignment_url"] == QCAA_ALIGNMENT_URL
        and QCAA_URL in crosswalk
        and QCAA_ALIGNMENT_URL in crosswalk,
        "Queensland Prep link pin drift",
    )
    need(
        snapshot["conditional_actual_system_code"] == "AC9TDIFK01"
        and set(snapshot["partial_paper_and_category_codes"])
        == {"AC9TDIFK02", "AC9TDIFP01"},
        "Partial/conditional claim accounting drift",
    )
    records = {
        record["code"]: record
        for record in imported["records"]
        if record.get("record_type") == "content_description"
        and record.get("code") in ROWS
    }
    need(set(records) == set(ROWS), "Official Foundation records missing")
    for code, row in ROWS.items():
        record = records[code]
        attrs = record["attributes"]
        need(
            record["source_row"] == row
            and attrs["level"] == "Foundation Year"
            and attrs["learning_area"] == "Technologies"
            and attrs["subject"] == "Digital Technologies",
            f"Wrong official level, area or row for {code}",
        )
        exact = (
            rf"^\| {code} \| {row} \| Technologies · Digital Technologies · "
            rf"Foundation Year \| {re.escape(record['plain_text'])} \|"
        )
        need(re.search(exact, crosswalk, re.MULTILINE), f"Verbatim row absent: {code}")
    need(
        "not observed" in crosswalk.lower() and "not secure exams" in crosswalk.lower(),
        "Partial/evidence boundary absent",
    )
    print("PASS exact three ACARA v9 Foundation rows and QCAA Prep source pins")


def pedagogy() -> None:
    """Confirm ten 25-minute scripts, three daily routes and twenty worked swaps."""
    lessons = read("LESSONS.md")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", lessons, re.MULTILINE))
    need([int(m.group(2)) for m in marks] == list(range(1, 11)), "Day sequence drift")
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        end = marks[index + 1].start() if index + 1 < len(marks) else len(lessons)
        body = lessons[mark.end() : end]
        times = [
            int(x)
            for x in re.findall(
                r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*",
                body,
                re.MULTILINE,
            )
        ]
        need(
            int(mark.group(1)) == (1 if day <= WEEK_SPLIT_DAY else 2)
            and times == [3, 4, 5, 7, 4, 2]
            and sum(times) == LESSON_MINUTES,
            f"Day {day} time/week drift: {times}",
        )
        target = re.search(r"^\*\*Target/codes:\*\* ([^\n]+)", body, re.MULTILINE)
        need(target, f"Day {day} target absent")
        codes = set(re.findall(r"\bAC9TDIF(?:K|P)\d\d\b", target.group(1)))
        need(codes == EXPECTED_CODES[day], f"Day {day} code drift: {codes}")
        need(
            "**Check/respond" in body
            and any(word in body.lower() for word in ("record", "capture")),
            f"Day {day} response evidence absent",
        )
    cards = read("LEARNER-CARDS.md")
    child_marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need(
        [int(m.group(1)) for m in child_marks] == list(range(1, 11)),
        "Learner card sequence drift",
    )
    for index, mark in enumerate(child_marks):
        end = (
            child_marks[index + 1].start()
            if index + 1 < len(child_marks)
            else len(cards)
        )
        body = cards[mark.end() : end]
        need(
            all(re.search(rf"^- \*\*{route} ·", body, re.MULTILINE) for route in "ABC")
            and "**Shared target:**" in body
            and "**Home:**" in body,
            f"Day {mark.group(1)} access routes/shared target/home absent",
        )
    swaps = read("PRACTICE-SWAPS.md")
    lines = re.findall(r"^\| (\d+) \| (.+) \| (.+) \|$", swaps, re.MULTILINE)
    need(
        [int(day) for day, _, _ in lines] == list(range(1, 11))
        and all(one.startswith("**") and two.startswith("**") for _, one, two in lines),
        "Twenty separate worked context swaps absent",
    )
    need(
        "not fixed learning styles" in cards
        and "no device" in cards.lower()
        and "not observed" in lessons.lower(),
        "Access or honest digital-evidence boundary absent",
    )
    print("PASS ten 25-minute scripts, 30 daily routes and 20 worked swaps")


def checks() -> None:
    """Check fresh printed cases, answer arithmetic and public key separation."""
    student = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    routine = read("PRACTICE-SWAPS.md") + read("LEARNER-CARDS.md")
    need(
        re.findall(r"^## Check ([AB]) · Day (\d+)\b", student, re.MULTILINE)
        == [("A", "5"), ("B", "10")],
        "Two held check days absent or reordered",
    )
    need(
        "teacher/KEY-AND-NEXT.md" not in student
        and "teacher/KEY-AND-NEXT.md" not in read("LEARNER-CARDS.md")
        and "public by url" in key.lower()
        and "not secure exams" in student.lower(),
        "Public-key/learner-copy boundary drift",
    )
    for leaked in (
        "HAT",
        "FLAG",
        "ROUND TOKEN",
        "Paper Costume Rack",
        "Paper Parade Box",
    ):
        need(leaked not in routine, f"Fresh source leaked into routine: {leaked}")
    for exact in (
        "HAT, HAT, BAG",
        "two HAT",
        "one BAG",
        "H H B",
        "MY NAME",
        "FLAG, ROUND TOKEN, ROUND TOKEN",
        "one FLAG",
        "two ROUND TOKEN",
        "F R R",
        "second position",
        "MY FACE PHOTO",
    ):
        need(exact in key, f"Fresh worked answer drift: {exact}")
    need(
        "no real" in student.lower()
        and "actual digital exploration is not observed" in key.lower(),
        "Privacy/system-evidence hold absent",
    )
    print("PASS two fresh public formative checks and separate worked key")


def svg_values(stem: str, attribute: str) -> Counter[str]:
    """Read machine-checkable geometry/source values from original SVG."""
    top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
    ns = {"svg": "http://www.w3.org/2000/svg"}
    return Counter(
        node.get(attribute)
        for node in top.findall(".//svg:rect", ns)
        if node.get(attribute) is not None
    )


def svg_source_order(stem: str) -> list[tuple[str, str]]:
    """Return exact indexed source entries in printed document order."""
    top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
    ns = {"svg": "http://www.w3.org/2000/svg"}
    return [
        (node.get("data-source-index", ""), node.get("data-entry", ""))
        for node in top.findall(".//svg:rect", ns)
        if node.get("data-entry") is not None
    ]


def assets() -> None:
    """Inspect generator inventory, exact source geometry and selectable A4 text."""
    alternatives = read("print/TEXT-ALTERNATIVES.md")
    tree = ast.parse(read("print/generate_print.py"))
    pages = next(
        node.value
        for node in tree.body
        if isinstance(node, ast.Assign)
        and any(
            isinstance(target, ast.Name) and target.id == "PAGES"
            for target in node.targets
        )
    )
    need(
        isinstance(pages, ast.Dict)
        and {key.value for key in pages.keys if isinstance(key, ast.Constant)}
        == set(STEMS),
        "Print generator page inventory drift",
    )
    ns = {"svg": "http://www.w3.org/2000/svg"}
    linear_alt = " ".join(alternatives.lower().split())
    for stem in STEMS:
        top = ET.parse(ROOT / "print" / f"{stem}.svg").getroot()
        pdf = ROOT / "print" / f"{stem}.pdf"
        need(
            top.get("width") == "210mm"
            and top.get("height") == "297mm"
            and top.find("svg:title", ns) is not None
            and top.find("svg:desc", ns) is not None,
            f"A4 SVG/metadata drift: {stem}",
        )
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        content = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        linear_pdf = " ".join(content.lower().split())
        need(
            re.search(r"Pages:\s+1\b", info)
            and "(A4)" in info
            and "CC BY 4.0" in content
            and len(content.split()) >= MIN_PDF_WORDS
            and f"{stem}.pdf" in alternatives
            and f"{stem}.svg" in alternatives,
            f"PDF/selectable-text/full alternative drift: {stem}",
        )
        for phrase in PRINTED_PHRASES.get(stem, ()):
            need(
                phrase.lower() in linear_pdf and phrase.lower() in linear_alt,
                f"Print/text source mismatch: {stem}: {phrase}",
            )
    need(
        svg_values("fictional-shelf-source", "data-entry")
        == Counter({"BOOK": 2, "CUP": 1})
        and svg_values("check-a-fresh-rack", "data-entry")
        == Counter({"HAT": 2, "BAG": 1})
        and svg_values("check-b-fresh-props", "data-entry")
        == Counter({"ROUND TOKEN": 2, "FLAG": 1}),
        "Printed source data/count drift",
    )
    for stem in (
        "fictional-shelf-source",
        "check-a-fresh-rack",
        "check-b-fresh-props",
    ):
        need(
            svg_values(stem, "data-source-index") == Counter({"1": 1, "2": 1, "3": 1}),
            f"Source order slots drift: {stem}",
        )
    need(
        svg_source_order("fictional-shelf-source")
        == [("1", "BOOK"), ("2", "BOOK"), ("3", "CUP")]
        and svg_source_order("check-a-fresh-rack")
        == [("1", "HAT"), ("2", "HAT"), ("3", "BAG")]
        and svg_source_order("check-b-fresh-props")
        == [("1", "FLAG"), ("2", "ROUND TOKEN"), ("3", "ROUND TOKEN")],
        "Printed indexed source order drift",
    )
    need(
        svg_values("check-b-fresh-props", "data-puppet-copy") == Counter({"F F R": 1})
        and svg_values("hardware-software-purpose", "data-part")
        == Counter({"SCREEN": 1, "KEYBOARD": 1, "SOFTWARE": 1}),
        "Puppet copy or hardware/software labels drift",
    )
    need(
        svg_values("check-a-fresh-rack", "data-category")
        == Counter({"MY NAME": 1, "PLAIN CIRCLE ICON": 1})
        and svg_values("check-b-fresh-props", "data-category")
        == Counter({"MY FACE PHOTO": 1, "PLAIN TRIANGLE ICON": 1}),
        "Fresh personal/neutral category cards drift",
    )
    need(
        "untagged" in alternatives and "tactile" in alternatives.lower(),
        "Print limit absent",
    )
    print("PASS eight A4 SVG/PDF pairs, selectable text and source geometry")


def interactive() -> None:
    """Fail on removed paper parity, keyboard semantics or data leakage APIs."""
    html = read("interactive/data-switch.html")
    paper = read("interactive/PAPER-EQUIVALENT.md")
    need(
        html.count('type="button" data-view=') == VIEW_BUTTONS
        and 'aria-pressed="true"' in html
        and 'role="status"' in html
        and 'aria-live="polite"' in html
        and 'Object.freeze(["BOOK", "BOOK", "CUP"])' in html
        and 'button.addEventListener("click"' in html,
        "Offline switcher source/keyboard/status drift",
    )
    need(
        not re.search(
            r"fetch\s*\(|XMLHttpRequest|localStorage|sessionStorage|"
            r"<form\b|<input\b|https?://|<script\s+src=",
            html,
            re.IGNORECASE,
        ),
        "Interactive includes prohibited network/input/storage surface",
    )
    need(
        "BOOK, BOOK, CUP" in paper
        and "three broad physical tokens" in paper.lower()
        and "AC9TDIFK02" in paper
        and "AC9TDIFK01" in paper
        and "not observed" in paper.lower(),
        "Full paper equivalent/evidence boundary drift",
    )
    local_hrefs = re.findall(r'<a\s+href="([^"]+)"', html)
    need(local_hrefs == ["PAPER-EQUIVALENT.md"], "Interactive link inventory drift")
    need(
        all((ROOT / "interactive" / href).is_file() for href in local_hrefs),
        "Broken interactive local link",
    )
    print("PASS offline switcher static guard and full paper/text equivalent")


def slug(title: str) -> str:
    """Make a simple Markdown heading anchor for local link validation."""
    title = re.sub(r"\[[^]]+\]\([^)]+\)", "", title)
    title = re.sub(r"[*_]", "", title).strip().lower()
    title = re.sub(r"[^\w -]", "", title)
    return title.replace(" ", "-")


def links(*, allow_manifest_bootstrap: bool) -> int:
    """Resolve every relative Markdown link and optional heading anchor."""
    count = 0
    for path in ROOT.rglob("*.md"):
        body = path.read_text(encoding="utf-8")
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", body):
            if url.startswith(("https://", "http://", "mailto:")):
                continue
            name, marker, anchor = unquote(url).partition("#")
            target = (path.parent / name).resolve() if name else path
            if allow_manifest_bootstrap and target == ROOT / "manifest.json":
                count += 1
                continue
            need(
                target.is_file(),
                f"Broken local link {path.relative_to(ROOT)} -> {url}",
            )
            if marker and target.suffix.lower() == ".md":
                headings = [
                    slug(x)
                    for x in re.findall(
                        r"^#{1,6} (.+)$",
                        target.read_text(encoding="utf-8"),
                        re.MULTILINE,
                    )
                ]
                need(anchor in headings, f"Broken heading anchor: {url}")
            count += 1
    print(f"PASS {count} local file and heading links")
    return count


def manifest_data() -> dict:
    """Hash every authored file except the self-referential manifest."""
    files = sorted(
        path
        for path in ROOT.rglob("*")
        if path.is_file()
        and path.name != "manifest.json"
        and "__pycache__" not in path.parts
    )
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/digital-technologies/term-1/weeks-01-02",
        "created_at": "2026-09-29",
        "review_status": "author_desk_checked_pending_classroom_device_access_review",
        "curriculum_source_sha256": SOURCE_SHA,
        "curriculum_codes_partial": ["AC9TDIFK02", "AC9TDIFP01"],
        "curriculum_code_conditional_actual_system": "AC9TDIFK01",
        "files": {
            str(path.relative_to(ROOT)): {
                "sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
                "bytes": path.stat().st_size,
            }
            for path in files
        },
    }


def main() -> None:
    """Check pack and optionally freeze fresh SHA-256 receipt."""
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    source()
    pedagogy()
    checks()
    assets()
    interactive()
    links(allow_manifest_bootstrap=args.write_manifest)
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n",
            encoding="utf-8",
        )
        print(f"WROTE SHA-256 manifest for {len(expected['files'])} files")
    else:
        need(
            json.loads(manifest.read_text(encoding="utf-8")) == expected,
            "SHA-256 manifest missing or stale",
        )
        print(f"PASS SHA-256 manifest for {len(expected['files'])} files")
    print("PACK PASS · author desk QA only; no live child or classroom validation")


if __name__ == "__main__":
    main()
