#!/usr/bin/env python3
"""Read-only integrity checks for Foundation HPE Weeks 5–6; opt-in manifest write."""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

from make_printables import PAGES

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = {
    "AC9HPFP01",
    "AC9HPFP02",
    "AC9HPFP03",
    "AC9HPFP04",
    "AC9HPFP05",
    "AC9HPFP06",
    "AC9HPFM01",
    "AC9HPFM02",
    "AC9HPFM03",
    "AC9HPFM04",
}
TEACHING = (
    "README.md",
    "MATERIALS.md",
    "LESSONS.md",
    "LEARNER-CARDS.md",
    "OPTIONAL-PRACTICE.md",
    "FAMILY-OPTIONAL.md",
)


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def official_rows() -> dict[str, dict]:
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    imported = json.loads(
        (STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8")
    )
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == imported["source_sha256"]
        == SOURCE_SHA,
        "Pinned official workbook/import hash differs",
    )
    rows = {
        row["code"]: row
        for row in imported["records"]
        if row.get("record_type") == "content_description" and row.get("code") in CODES
    }
    text = (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8")
    table: dict[str, tuple[int, str]] = {}
    for line in text.splitlines():
        match = re.match(
            r"^\| (AC9HPF[PM]\d\d) \| (\d+) \| Health and Physical Education · Foundation Year \| (.*?) \|",
            line,
        )
        if match:
            table[match.group(1)] = (int(match.group(2)), match.group(3))
    need(
        set(rows) == set(table) == CODES,
        "Crosswalk does not contain all exact HPE rows",
    )
    for code, (source_row, description) in table.items():
        row = rows[code]
        need(
            row["attributes"]["learning_area"] == "Health and Physical Education"
            and row["attributes"]["level"] == "Foundation Year",
            f"Wrong source area/level: {code}",
        )
        need(
            row["source_row"] == source_row
            and " ".join(row["plain_text"].split()) == " ".join(description.split()),
            f"Incorrect exact content description: {code}",
        )
    return rows


def lessons(rows: dict[str, dict]) -> None:
    text = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", text, re.MULTILINE))
    need(
        [int(m.group(2)) for m in marks] == list(range(21, 31)),
        "Need Days 21–30 exactly",
    )
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        body = text[
            mark.end() : marks[index + 1].start()
            if index + 1 < len(marks)
            else len(text)
        ]
        need(int(mark.group(1)) == (day - 1) // 5 + 1, f"Day {day} wrong week")
        minutes = [
            int(n)
            for n in re.findall(r"^\d+\. \*\*[^\n]*?\b(\d+) min\.", body, re.MULTILINE)
        ]
        need(
            len(minutes) == 6 and sum(minutes) == 25,
            f"Day {day} timed script: {minutes}",
        )
        need("**Goal:**" in body and "**Prepare:**" in body, f"Day {day} missing setup")
        for code in re.findall(r"\bAC9[A-Z0-9]+\b", body):
            need(code in rows, f"Day {day} uses unverified code {code}")
    need(
        "OUTDOOR PHYSICAL ACTIVITY NOT OBSERVED" in text
        and "PHYSICAL ACTIVITY NOT OBSERVED" in text
        and "first independent" in text.lower()
        and "school policy" in text,
        "Evidence/safeguarding boundary missing in lessons",
    )


def routes_swaps_and_checks() -> None:
    cards = (ROOT / "LEARNER-CARDS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need(
        [int(m.group(1)) for m in marks] == list(range(21, 31)),
        "Ten route days missing",
    )
    for i, mark in enumerate(marks):
        section = cards[
            mark.end() : marks[i + 1].start() if i + 1 < len(marks) else len(cards)
        ]
        need(
            sorted(re.findall(r"^- \*\*([ABC]): ", section, re.MULTILINE))
            == ["A", "B", "C"],
            f"Day {mark.group(1)} needs A/B/C same-target routes",
        )
    swaps = (ROOT / "OPTIONAL-PRACTICE.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", swaps, re.MULTILINE))
    need(
        [int(m.group(1)) for m in marks] == list(range(21, 31)),
        "Ten practice days missing",
    )
    for i, mark in enumerate(marks):
        section = swaps[
            mark.end() : marks[i + 1].start() if i + 1 < len(marks) else len(swaps)
        ]
        need(
            len(
                re.findall(
                    r"^[12]\. \*\*.+?\*\* .*?\*\*Worked report:\*\* ",
                    section,
                    re.MULTILINE,
                )
            )
            == 2,
            f"Day {mark.group(1)} needs two worked swaps",
        )
    family = (ROOT / "FAMILY-OPTIONAL.md").read_text(encoding="utf-8")
    need(
        [int(x) for x in re.findall(r"^\| (\d\d) \|", family, re.MULTILINE)]
        == list(range(21, 31)),
        "Ten optional family bridges missing",
    )
    need(
        "not homework" in family.lower() and "no family needs to buy" in family.lower(),
        "Family burden boundary absent",
    )

    lesson = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    assessment = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    # Exact day/case pairing across all three files closes the prior Science W23–24 mismatch.
    lesson_cases = re.findall(
        r"^## Week \d+ · Day (\d+)[^\n]*\n\n\*\*Goal:\*\*[^\n]*?\[teacher-held Case ([A-Z]+)\]\(ASSESSMENT\.md\)",
        lesson,
        re.MULTILINE,
    )
    assessment_cases = re.findall(
        r"^## Day (\d+) · fresh Case ([A-Z]+)", assessment, re.MULTILINE
    )
    key_cases = re.findall(r"^## Day (\d+) · Case ([A-Z]+)", key, re.MULTILINE)
    expected = [("25", "H"), ("30", "I")]
    need(
        lesson_cases == assessment_cases == key_cases == expected,
        f"Held case/day IDs differ: lessons={lesson_cases}, assessment={assessment_cases}, key={key_cases}",
    )
    teaching = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in TEACHING)
    for signature in (
        "BRONZE LOOP MAP",
        "COPPER COMET BOARD",
        "QUIET CORNER",
        "WAIT ARC",
    ):
        need(signature in assessment, f"Fresh source signature absent: {signature}")
        need(
            signature.lower() not in teaching.lower(),
            f"Held case leaked into teaching: {signature}",
        )
    need(
        "first independent" in assessment.lower()
        and "OUTDOOR PHYSICAL ACTIVITY NOT OBSERVED" in key
        and "PHYSICAL ACTIVITY NOT OBSERVED" in key
        and "school policy" in key,
        "First-response, real observation or safeguarding boundary missing",
    )
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8")
    need(
        all(
            token in materials
            for token in (
                "OUTDOOR PHYSICAL ACTIVITY NOT OBSERVED",
                "PHYSICAL ACTIVITY NOT OBSERVED",
                "Source A",
                "Source E",
                "tactile",
            )
        ),
        "Outdoor/physical gate, source text or tactile alternatives missing",
    )


def assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    need(len(PAGES) == 6, "Six aids required")
    for stem, _title, _subtitle, labels in PAGES:
        svg = ROOT / f"{stem}.svg"
        pdf = ROOT / f"{stem}.pdf"
        tree = ET.parse(svg).getroot()
        need(
            tree.get("width") == "210mm" and tree.get("height") == "297mm",
            f"{stem} not A4 SVG",
        )
        desc = tree.find("s:desc", ns)
        need(
            desc is not None and all(label in (desc.text or "") for label in labels),
            f"{stem} SVG exact text/desc missing",
        )
        pdf_info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        need(
            "Pages:           1" in pdf_info and "(A4)" in pdf_info,
            f"{stem} not one-page A4 PDF",
        )
        pdf_text = " ".join(
            subprocess.check_output(["pdftotext", str(pdf), "-"], text=True).split()
        )
        need(
            all(label in pdf_text for label in labels) and "CC BY 4.0" in pdf_text,
            f"{stem} PDF exact field or licence text missing",
        )


def local_links() -> int:
    count = 0
    for path in ROOT.glob("*.md"):
        for url in re.findall(
            r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")
        ):
            if url.startswith(("https://", "http://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            need(target.exists(), f"Broken local link: {path.name} → {url}")
            count += 1
    return count


def markdown_list_spacing() -> int:
    count = 0
    marker = re.compile(r"^(?:- |\d+\. )")
    for path in ROOT.glob("*.md"):
        lines = path.read_text(encoding="utf-8").splitlines()
        for index, line in enumerate(lines):
            if not marker.match(line):
                continue
            count += 1
            previous = lines[index - 1] if index else ""
            following = lines[index + 1] if index + 1 < len(lines) else ""
            need(
                not previous.strip() or bool(marker.match(previous)),
                f"List needs blank line above: {path.name}:{index + 1}",
            )
            need(
                not following.strip() or bool(marker.match(following)),
                f"List needs blank line below: {path.name}:{index + 1}",
            )
    return count


def manifest_data() -> dict:
    files = sorted(
        path
        for path in ROOT.rglob("*")
        if path.is_file()
        and path.name != "manifest.json"
        and "__pycache__" not in path.parts
        and ".ruff_cache" not in path.parts
    )
    return {
        "pack_id": "subjectnest-acara-v9-foundation-hpe-t1-w05-06",
        "pack_version": "0.1.0-draft",
        "created_at": "2026-09-30",
        "review_status": "pending_human_educator_safeguarding_accessibility_local_syllabus_equipment_and_classroom_review",
        "year_label": "Australian Curriculum Foundation Year / Queensland Prep",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(CODES),
        "curriculum_source": {
            "authority": "Australian Curriculum, Assessment and Reporting Authority",
            "source_url": "https://www.australiancurriculum.edu.au/downloads",
            "retrieved_at": "2026-09-29",
            "source_hash": SOURCE_SHA,
            "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use",
        },
        "rights": "Original SubjectNest scripts, stories, checks and vector aids CC BY 4.0; ACARA and QCAA retain their own terms.",
        "original_assets": [
            {
                "item_code_or_locator": str(path.relative_to(ROOT)),
                "sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
            }
            for path in files
        ],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    rows = official_rows()
    lessons(rows)
    routes_swaps_and_checks()
    assets()
    link_count = local_links()
    list_count = markdown_list_spacing()
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
        )
    need(
        json.loads(manifest.read_text(encoding="utf-8")) == expected,
        "Hash manifest stale or missing",
    )
    print(
        f"PASS HPE W5–6: 10x25m, 30 routes, 20 swaps, 10 family bridges, 2 matched held cases, 10 exact rows, 6 A4 aids, {link_count} links, {list_count} spaced list items, hashes"
    )


if __name__ == "__main__":
    main()
