#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Verify Foundation Science Weeks 19–20 against the pinned official source."""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = {
    "AC9SFU02",
    "AC9SFH01",
    "AC9SFI01",
    "AC9SFI02",
    "AC9SFI03",
    "AC9SFI04",
    "AC9SFI05",
}
STEMS = (
    "question-predict",
    "setup-change",
    "four-run",
    "claim-limit",
    "source-portfolio",
    "handover",
)
TEACHING = (
    "README.md",
    "LESSONS.md",
    "MATERIALS.md",
    "LEARNER-CARDS.md",
    "OPTIONAL-PRACTICE.md",
)


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def official() -> dict[str, dict]:
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    data = json.loads(
        (STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8")
    )
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == SOURCE_SHA
        == data["source_sha256"],
        "Official workbook/import hash differs",
    )
    rows = {
        r["code"]: r
        for r in data["records"]
        if r.get("record_type") == "content_description" and r.get("code") in CODES
    }
    table: dict[str, tuple[int, str]] = {}
    for line in (
        (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8").splitlines()
    ):
        match = re.match(
            r"^\| (AC9[A-Z0-9]+) \| (\d+) \| Science · Foundation Year \| (.*?) \|$",
            line,
        )
        if match:
            table[match.group(1)] = (int(match.group(2)), match.group(3))
    need(set(table) == CODES == set(rows), "Crosswalk/source code mismatch")
    for code, (source_row, description) in table.items():
        row = rows[code]
        need(
            row["attributes"]["level"] == "Foundation Year"
            and row["attributes"]["learning_area"] == "Science",
            f"Wrong level/area: {code}",
        )
        need(
            source_row == row["source_row"]
            and " ".join(description.split()) == " ".join(row["plain_text"].split()),
            f"Wrong source row/text: {code}",
        )
    return rows


def lessons(rows: dict[str, dict]) -> None:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", content, re.MULTILINE))
    need(
        [int(m.group(2)) for m in marks] == list(range(91, 101)),
        "Ten Science days 91–100 required",
    )
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (day - 1) // 5 + 1, f"Wrong week for Day {day}")
        body = content[
            mark.end() : marks[index + 1].start()
            if index + 1 < len(marks)
            else len(content)
        ]
        times = [
            int(value)
            for value in re.findall(
                r"^\d+\. \*\*[^\n]*?\b(\d+) min\.", body, re.MULTILINE
            )
        ]
        need(
            len(times) == 6 and sum(times) == 25,
            f"Day {day} needs six 25-minute steps: {times}",
        )
        need(
            "**Goal:**" in body and "**Prepare:**" in body,
            f"Day {day} goal/preparation missing",
        )
        for code in set(re.findall(r"\bAC9[A-Z0-9]+\b", body)):
            need(code in rows, f"Unknown/non-Foundation lesson code: {code}")
    need(
        "first independent" in content.lower()
        and "REAL OBSERVATION NOT AVAILABLE" in content,
        "Assessment/source boundary absent",
    )


def cards_practice_checks() -> None:
    cards = (ROOT / "LEARNER-CARDS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need(
        [int(m.group(1)) for m in marks] == list(range(91, 101)),
        "Ten learner-card days required",
    )
    for index, mark in enumerate(marks):
        body = cards[
            mark.end() : marks[index + 1].start()
            if index + 1 < len(marks)
            else len(cards)
        ]
        need(
            len(re.findall(r"^- \*\*[^*]+:\*\* .+$", body, re.MULTILINE)) == 3,
            f"Day {mark.group(1)} needs three routes",
        )
    need(
        "ASSESSMENT.md" not in cards and "TEACHER-KEY.md" not in cards,
        "Held content linked from learner page",
    )

    practice = (ROOT / "OPTIONAL-PRACTICE.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", practice, re.MULTILINE))
    need(
        [int(m.group(1)) for m in marks] == list(range(91, 101)),
        "Ten worked-swap days required",
    )
    for index, mark in enumerate(marks):
        body = practice[
            mark.end() : marks[index + 1].start()
            if index + 1 < len(marks)
            else len(practice)
        ]
        need(
            len(
                re.findall(
                    r"^[12]\. \*\*.+?\*\* .+?\*\*Worked report:\*\* .+$",
                    body,
                    re.MULTILINE,
                )
            )
            == 2,
            f"Day {mark.group(1)} needs two worked swaps",
        )

    assessment = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    teaching = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in TEACHING)
    for phrase in (
        "HOME, BRIDGE, GATE",
        "fabric-covered block",
        "MARKET DISPLAY PANEL MODEL",
        "HAND CAUGHT PANEL",
    ):
        need(
            phrase.lower() not in teaching.lower(),
            f"Held case leaked into teaching: {phrase}",
        )
        need(phrase.lower() in assessment.lower(), f"Held case missing: {phrase}")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    need(
        all(f"Day {day}" in assessment and f"Day {day}" in key for day in (95, 100)),
        "Two held checks/separate keys required",
    )
    need(
        "first independent" in assessment.lower()
        and "REAL OBSERVATION NOT AVAILABLE" in key,
        "Fresh check/source gate absent",
    )
    need(
        "no fixed local movement answer" in key.lower()
        and "not the child's own direct sense" in key.lower(),
        "Key overcredits real or reported evidence",
    )
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8")
    need(
        all(
            term in materials
            for term in (
                "local date",
                "approved",
                "REAL OBSERVATION NOT AVAILABLE",
                "ADULT REPORT",
                "tactile",
                "FICTIONAL MODEL",
            )
        ),
        "Core source/access/control facts missing",
    )
    family = (ROOT / "FAMILY-OPTIONAL.md").read_text(encoding="utf-8")
    need(
        [int(d) for d in re.findall(r"^\| (\d{2,3}) \|", family, re.MULTILINE)]
        == list(range(91, 101)),
        "Ten optional family prompts required",
    )
    need(
        "no family needs to buy" in family.lower() and "not homework" in family.lower(),
        "Family boundary absent",
    )


def assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    labels = {
        "question-predict": (
            "SOURCE + DATE / FICTIONAL MODEL",
            "MY PREDICTION BEFORE ANY RESULT",
        ),
        "setup-change": (
            "LARGE CLOSED CARTON + RIMMED TRAY",
            "SAME TRAY / START / RELEASE",
        ),
        "four-run": (
            "ACTUAL SOURCE + LOCAL DATE / FICTIONAL MODEL",
            "RUN 4 FACE + STOP / INTERRUPTED",
        ),
        "claim-limit": (
            "VALID / INTERRUPTED COUNTS",
            "CLAIM ABOUT THESE TRIES ONLY",
        ),
        "source-portfolio": ("SAMPLE A SOURCE + DATE", "PRIVATE AUDIENCE CHOICE"),
        "handover": (
            "SOURCE + DATE OR MODEL",
            "ACTUAL LISTENER RESPONSE OR NONE",
        ),
    }
    for stem in STEMS:
        svg, pdf = ROOT / f"{stem}.svg", ROOT / f"{stem}.pdf"
        tree = ET.parse(svg).getroot()
        need(
            tree.attrib.get("width") == "210mm"
            and tree.attrib.get("height") == "297mm",
            f"{stem} SVG is not A4",
        )
        need(
            tree.find("s:title", ns) is not None
            and tree.find("s:desc", ns) is not None,
            f"{stem} lacks SVG title/desc",
        )
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        need(
            "Pages:           1" in info and "(A4)" in info,
            f"{stem} PDF is not single A4",
        )
        extracted = " ".join(
            subprocess.check_output(["pdftotext", str(pdf), "-"], text=True).split()
        )
        need(
            all(label in extracted for label in labels[stem])
            and "CC BY 4.0" in extracted,
            f"{stem} PDF text/credit missing",
        )


def links() -> int:
    count = 0
    for path in ROOT.glob("*.md"):
        for url in re.findall(
            r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")
        ):
            if url.startswith(("https://", "http://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            need(target.exists(), f"Broken local link: {path.name} -> {url}")
            count += 1
    return count


def manifest_data() -> dict:
    files = sorted(
        path
        for path in ROOT.rglob("*")
        if path.is_file()
        and path.name != "manifest.json"
        and "__pycache__" not in path.parts
        and not path.name.startswith("_preview")
    )
    return {
        "pack_id": "subjectnest-acara-v9-foundation-science-t2-w19-20",
        "pack_version": "0.1.0-draft",
        "created_at": "2026-09-29",
        "review_status": "pending_human_educator_accessibility_local_syllabus_safety_privacy_and_classroom_review",
        "year_label": "Australian Curriculum Foundation Year / Queensland Prep",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(CODES),
        "curriculum_source": {
            "authority": "Australian Curriculum, Assessment and Reporting Authority",
            "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx",
            "retrieved_at": "2026-09-29",
            "source_hash": SOURCE_SHA,
            "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use",
        },
        "rights": "Original SubjectNest texts, scripts, models and vector aids CC BY 4.0; ACARA retains its own terms.",
        "original_assets": [
            {
                "item_code_or_locator": str(path.relative_to(ROOT)),
                "sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
            }
            for path in files
        ],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    rows = official()
    lessons(rows)
    cards_practice_checks()
    assets()
    link_count = links()
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
        )
    need(
        json.loads(manifest.read_text(encoding="utf-8")) == expected,
        "Hash manifest stale or missing",
    )
    print(
        f"PASS Foundation Science W19–20: 10x25m, 30 routes, 20 swaps, 10 family prompts, 2 fresh checks/keys, 7 exact codes, 6 A4 aids, {link_count} local links, hashes"
    )


if __name__ == "__main__":
    main()
