#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Check Foundation Science Weeks 9–10 against pinned ACARA rows and local assets."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = {"AC9SFU01", "AC9SFU02", "AC9SFU03", "AC9SFH01", "AC9SFI01", "AC9SFI02", "AC9SFI03", "AC9SFI04", "AC9SFI05"}
TEACHING = ("README.md", "LESSONS.md", "MATERIALS.md", "LEARNER-CARDS.md", "OPTIONAL-PRACTICE.md")
STEMS = ("source-gate", "evidence-display", "compare-records", "fictional-notebook", "next-question", "handover")


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def official() -> dict[str, dict]:
    workbook = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    data = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8"))
    need(hashlib.sha256(workbook.read_bytes()).hexdigest() == SOURCE_SHA == data["source_sha256"], "Official workbook/import hash differs")
    rows = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description" and r.get("code") in CODES}
    table: dict[str, tuple[int, str]] = {}
    for line in (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8").splitlines():
        match = re.match(r"^\| (AC9[A-Z0-9]+) \| (\d+) \| Science · Foundation Year \| (.*?) \|$", line)
        if match:
            table[match.group(1)] = (int(match.group(2)), match.group(3))
    need(set(table) == CODES == set(rows), "Crosswalk/source code mismatch")
    for code, (source_row, description) in table.items():
        row = rows[code]
        need(row["attributes"]["level"] == "Foundation Year" and row["attributes"]["learning_area"] == "Science", f"Wrong level/area: {code}")
        need(source_row == row["source_row"] and " ".join(description.split()) == " ".join(row["plain_text"].split()), f"Wrong source row/text: {code}")
    return rows


def lessons(rows: dict[str, dict]) -> None:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", content, re.MULTILINE))
    need([int(m.group(2)) for m in marks] == list(range(41, 51)), "Ten Science days 41–50 required")
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (day - 1) // 5 + 1, f"Wrong week for Day {day}")
        body = content[mark.end():marks[index + 1].start() if index + 1 < len(marks) else len(content)]
        times = [int(value) for value in re.findall(r"^\d+\. \*\*[^\n]*?\b(\d+) min\.", body, re.MULTILINE)]
        need(len(times) == 6 and sum(times) == 25, f"Day {day} needs six 25-minute steps: {times}")
        need("**Goal:**" in body and "**Prepare:**" in body, f"Day {day} goal/preparation missing")
        for code in set(re.findall(r"\bAC9[A-Z0-9]+\b", body)):
            need(code in rows, f"Unknown/non-Foundation lesson code: {code}")
    need("first independent" in content.lower() and "REAL OBSERVATION NOT AVAILABLE" in content, "Assessment/source boundary absent")


def cards_practice_checks() -> None:
    cards = (ROOT / "LEARNER-CARDS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need([int(m.group(1)) for m in marks] == list(range(41, 51)), "Ten learner-card days required")
    for index, mark in enumerate(marks):
        body = cards[mark.end():marks[index + 1].start() if index + 1 < len(marks) else len(cards)]
        routes = re.findall(r"^- \*\*[^*]+:\*\* .+$", body, re.MULTILINE)
        need(len(routes) == 3, f"Day {mark.group(1)} needs three routes")
    need("ASSESSMENT.md" not in cards and "TEACHER-KEY.md" not in cards, "Held content linked from learner page")

    practice = (ROOT / "OPTIONAL-PRACTICE.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", practice, re.MULTILINE))
    need([int(m.group(1)) for m in marks] == list(range(41, 51)), "Ten worked-swap days required")
    for index, mark in enumerate(marks):
        body = practice[mark.end():marks[index + 1].start() if index + 1 < len(marks) else len(practice)]
        swaps = re.findall(r"^[12]\. \*\*.+?\*\* .+?\*\*Worked(?: report)?:\*\* .+$", body, re.MULTILINE)
        need(len(swaps) == 2, f"Day {mark.group(1)} needs two worked swaps")

    assessment = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    teaching = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in TEACHING)
    for phrase in ("stepped outline", "two large circular outlines", "kite-shaped outline", "three short horizontal stripes", "always slide"):
        need(phrase.lower() not in teaching.lower(), f"Held case leaked into teaching: {phrase}")
        need(phrase.lower() in assessment.lower(), f"Held case missing: {phrase}")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    need(all(f"Day {day}" in assessment and f"Day {day}" in key for day in (45, 50)), "Two held checks/separate keys required")
    need("first independent" in assessment.lower() and "REAL OBSERVATION NOT AVAILABLE" in assessment, "Fresh check/source gate absent")
    need("no fixed real-object or movement answer" in key.lower() and "not the child's direct sense" in key.lower(), "Key overcredits real or reported evidence")
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8")
    need(all(term in materials for term in ("local date", "verified", "REAL OBSERVATION NOT AVAILABLE", "ADULT REPORT", "tactile", "NEW EVIDENCE".lower())), "Core source/access/control facts missing")
    family = (ROOT / "FAMILY-OPTIONAL.md").read_text(encoding="utf-8")
    need([int(d) for d in re.findall(r"^\| (4\d|50) \|", family, re.MULTILINE)] == list(range(41, 51)), "Ten optional family prompts required")
    need("no family needs to buy" in family.lower() and "not homework" in family.lower(), "Family boundary absent")


def assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    labels = {
        "source-gate": ("1 SOURCE", "2 DATE", "3 ACCESS", "4 ACTION", "5 RESULT", "6 LIMIT"),
        "evidence-display": ("SOURCE + DATE + ACCESS", "MY PREDICTION BEFORE ANY TRY", "WHAT IS STILL UNKNOWN"),
        "compare-records": ("RECORD A", "RECORD B", "CAUTIOUS COMPARISON"),
        "fictional-notebook": ("INVENTED NOTEBOOK", "PREDICTION IN STORY", "STOP AT MIDDLE"),
        "next-question": ("ONE CLAIM I CAN SUPPORT", "MY NEXT SAFE QUESTION", "NOT TESTED"),
        "handover": ("WORK SAMPLE", "CHILD'S EXACT IDEA", "NEXT TEACHING MOVE"),
    }
    for stem in STEMS:
        svg, pdf = ROOT / f"{stem}.svg", ROOT / f"{stem}.pdf"
        tree = ET.parse(svg).getroot()
        need(tree.attrib.get("width") == "210mm" and tree.attrib.get("height") == "297mm", f"{stem} SVG is not A4")
        need(tree.find("s:title", ns) is not None and tree.find("s:desc", ns) is not None, f"{stem} lacks SVG title/desc")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        need("Pages:           1" in info and "(A4)" in info, f"{stem} PDF is not single A4")
        extracted = " ".join(subprocess.check_output(["pdftotext", str(pdf), "-"], text=True).split())
        need(all(label in extracted for label in labels[stem]) and "CC BY 4.0" in extracted, f"{stem} PDF text/credit missing")
    model = ET.parse(ROOT / "fictional-notebook.svg").getroot()
    need(len(model.findall("s:rect", ns)) >= 6, "Fictional notebook panels differ")


def links() -> int:
    count = 0
    for path in ROOT.glob("*.md"):
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("https://", "http://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            need(target.exists(), f"Broken local link: {path.name} -> {url}")
            count += 1
    return count


def manifest_data() -> dict:
    files = sorted(path for path in ROOT.rglob("*") if path.is_file() and path.name != "manifest.json" and "__pycache__" not in path.parts and not path.name.startswith("_preview"))
    return {
        "pack_id": "subjectnest-acara-v9-foundation-science-t1-w09-10",
        "pack_version": "0.1.0-draft",
        "created_at": "2026-09-29",
        "review_status": "pending_human_educator_accessibility_local_syllabus_safety_privacy_and_classroom_review",
        "year_label": "Australian Curriculum Foundation Year / Queensland Prep",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(CODES),
        "curriculum_source": {
            "authority": "Australian Curriculum, Assessment and Reporting Authority",
            "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx",
            "retrieved_at": "2026-09-29",
            "source_hash": SOURCE_SHA,
            "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use",
        },
        "rights": "Original SubjectNest texts, scripts, models and vector aids CC BY 4.0; ACARA retains its own terms.",
        "original_assets": [{"item_code_or_locator": str(path.relative_to(ROOT)), "sha256": hashlib.sha256(path.read_bytes()).hexdigest()} for path in files],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    rows = official()
    lessons(rows)
    cards_practice_checks()
    assets()
    link_count = links()
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
    need(json.loads(manifest.read_text(encoding="utf-8")) == expected, "Hash manifest stale or missing")
    print(f"PASS Foundation Science W09–10: 10x25m, 30 routes, 20 swaps, 10 family prompts, 2 fresh checks/keys, 9 exact codes, 6 A4 aids, {link_count} local links, hashes")


if __name__ == "__main__":
    main()
