#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only structural/source/asset verification for Science Weeks 25–26."""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = {
    "AC9SFU02",
    "AC9SFH01",
    "AC9SFI01",
    "AC9SFI02",
    "AC9SFI03",
    "AC9SFI04",
    "AC9SFI05",
}
STEMS = (
    "three-stills",
    "two-possible-routes",
    "safe-real-view",
    "path-observation",
    "raw-attempts",
    "claim-listener",
)
TEACHING = (
    "README.md",
    "LESSONS.md",
    "MATERIALS.md",
    "LEARNER-CARDS.md",
    "OPTIONAL-PRACTICE.md",
)


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def official() -> dict[str, dict]:
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    data = json.loads(
        (STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8")
    )
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == SOURCE_SHA
        == data["source_sha256"],
        "Official workbook/import hash differs",
    )
    rows = {
        r["code"]: r
        for r in data["records"]
        if r.get("record_type") == "content_description" and r.get("code") in CODES
    }
    table: dict[str, tuple[int, str]] = {}
    for line in (
        (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8").splitlines()
    ):
        match = re.match(
            r"^\| (AC9[A-Z0-9]+) \| (\d+) \| Science · Foundation Year \| (.*?) \|$",
            line,
        )
        if match:
            table[match.group(1)] = (int(match.group(2)), match.group(3))
    need(set(table) == CODES == set(rows), "Crosswalk/source code mismatch")
    for code, (source_row, description) in table.items():
        row = rows[code]
        need(
            row["attributes"]["level"] == "Foundation Year"
            and row["attributes"]["learning_area"] == "Science",
            f"Wrong level/area: {code}",
        )
        need(
            source_row == row["source_row"]
            and " ".join(description.split()) == " ".join(row["plain_text"].split()),
            f"Wrong source row/text: {code}",
        )
    return rows


def lessons(rows: dict[str, dict]) -> None:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", content, re.MULTILINE))
    need(
        [int(m.group(2)) for m in marks] == list(range(121, 131)),
        "Ten Science days 121–130 required",
    )
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (day - 1) // 5 + 1, f"Wrong week for Day {day}")
        body = content[
            mark.end() : marks[index + 1].start()
            if index + 1 < len(marks)
            else len(content)
        ]
        times = [
            int(value)
            for value in re.findall(
                r"^\d+\. \*\*[^\n]*?\b(\d+) min\.", body, re.MULTILINE
            )
        ]
        need(
            len(times) == 6 and sum(times) == 25,
            f"Day {day} needs six 25-minute steps: {times}",
        )
        need(
            "**Goal:**" in body and "**Prepare:**" in body,
            f"Day {day} goal/preparation missing",
        )
        for code in set(re.findall(r"\bAC9[A-Z0-9]+\b", body)):
            need(code in rows, f"Unknown/non-Foundation lesson code: {code}")
    need(
        "first independent" in content.lower()
        and "REAL OBSERVATION NOT AVAILABLE" in content,
        "Assessment/source boundary absent",
    )


def cards_practice_checks() -> None:
    cards = (ROOT / "LEARNER-CARDS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need(
        [int(m.group(1)) for m in marks] == list(range(121, 131)),
        "Ten learner-card days required",
    )
    for index, mark in enumerate(marks):
        body = cards[
            mark.end() : marks[index + 1].start()
            if index + 1 < len(marks)
            else len(cards)
        ]
        need(
            len(re.findall(r"^- \*\*[^*]+:\*\* .+$", body, re.MULTILINE)) == 3,
            f"Day {mark.group(1)} needs three same-goal routes",
        )
    need(
        "ASSESSMENT.md" not in cards and "TEACHER-KEY.md" not in cards,
        "Held content linked from learner page",
    )

    practice = (ROOT / "OPTIONAL-PRACTICE.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", practice, re.MULTILINE))
    need(
        [int(m.group(1)) for m in marks] == list(range(121, 131)),
        "Ten worked-swap days required",
    )
    for index, mark in enumerate(marks):
        body = practice[
            mark.end() : marks[index + 1].start()
            if index + 1 < len(marks)
            else len(practice)
        ]
        need(
            len(
                re.findall(
                    r"^[12]\. \*\*.+?\*\* .+?\*\*Worked report:\*\* .+$",
                    body,
                    re.MULTILINE,
                )
            )
            == 2,
            f"Day {mark.group(1)} needs two worked swaps",
        )

    assessment = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    teaching = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in TEACHING)
    for phrase in (
        "ORANGE RECTANGLE 46",
        "LOW LEFT",
        "GREEN SQUARE 73",
        "ROLLED THROUGH THE CENTRE",
    ):
        need(
            phrase.lower() not in teaching.lower(),
            f"Held case leaked into teaching: {phrase}",
        )
        need(phrase.lower() in assessment.lower(), f"Held case missing: {phrase}")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    lesson_cases = re.findall(
        r"\[teacher-held Case ([A-Z]+)\]\(ASSESSMENT\.md\)",
        (ROOT / "LESSONS.md").read_text(encoding="utf-8"),
    )
    assessment_cases = re.findall(
        r"^## Day \d+ · fresh Case ([A-Z]+)", assessment, re.MULTILINE
    )
    key_cases = re.findall(r"^## Day \d+ · Case ([A-Z]+)", key, re.MULTILINE)
    need(
        lesson_cases == assessment_cases == key_cases == ["AB", "AC"],
        "Held case IDs differ between lessons, cases and key",
    )
    need(
        all(f"Day {day}" in assessment and f"Day {day}" in key for day in (125, 130)),
        "Two held checks/separate keys required",
    )
    need(
        "first independent" in assessment.lower()
        and "REAL OBSERVATION NOT AVAILABLE" in key
        and "no fixed local movement answer" in key.lower()
        and "not the child's own direct sense" in key.lower(),
        "Fresh check or real-source boundary absent",
    )
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8")
    need(
        all(
            term in materials
            for term in (
                "local date",
                "approved",
                "REAL OBSERVATION NOT AVAILABLE",
                "ADULT REPORT",
                "tactile",
                "FICTIONAL MODEL",
                "SOURCE",
            )
        ),
        "Core source/access/safety facts missing",
    )
    family = (ROOT / "FAMILY-OPTIONAL.md").read_text(encoding="utf-8")
    need(
        [int(d) for d in re.findall(r"^\| (\d{2,3}) \|", family, re.MULTILINE)]
        == list(range(121, 131)),
        "Ten optional family prompts required",
    )
    need(
        "no family needs to buy" in family.lower() and "not homework" in family.lower(),
        "Family boundary absent",
    )
    read_aloud = (ROOT / "READ-ALOUD.md").read_text(encoding="utf-8")
    need(
        len(re.findall(r"^## Track [123] ·", read_aloud, re.MULTILINE)) == 3
        and "no recorded audio" in read_aloud.lower()
        and "fictional model" in read_aloud.lower(),
        "Three source-labelled narration text masters or audio status missing",
    )


def assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    labels = {
        "three-stills": ("POSITIONS SHOWN", "PATH BETWEEN UNKNOWN", "PAUSE UNKNOWN"),
        "two-possible-routes": (
            "SOLID POSSIBLE ROUTE",
            "DASHED POSSIBLE ROUTE",
            "BOTH FICTIONAL",
        ),
        "safe-real-view": (
            "APPROVAL + LOCAL DATE",
            "ADULT MOVES SLOWLY",
            "STOP IF INTERRUPTED",
        ),
        "path-observation": (
            "PRIOR PREDICTION OR NONE",
            "PATH SEEN / PART HIDDEN",
            "DIRECT / ADULT REPORT / MODEL",
        ),
        "raw-attempts": ("TRY 1", "TRY 2", "NO FIXED RESULT"),
        "claim-listener": (
            "WHAT SOURCE SHOWS",
            "UNKNOWN / HIDDEN PART",
            "PRIVATE LISTENER RESPONSE OR NONE",
        ),
    }
    for stem in STEMS:
        svg, pdf = ROOT / f"{stem}.svg", ROOT / f"{stem}.pdf"
        tree = ET.parse(svg).getroot()
        need(
            tree.attrib.get("width") == "210mm"
            and tree.attrib.get("height") == "297mm",
            f"{stem} SVG is not A4",
        )
        need(
            tree.find("s:title", ns) is not None
            and tree.find("s:desc", ns) is not None,
            f"{stem} lacks SVG title/desc",
        )
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        need(
            "Pages:           1" in info and "(A4)" in info,
            f"{stem} PDF is not single A4",
        )
        extracted = " ".join(
            subprocess.check_output(["pdftotext", str(pdf), "-"], text=True).split()
        )
        need(
            all(label in extracted for label in labels[stem])
            and "CC BY 4.0" in extracted,
            f"{stem} PDF text/credit missing",
        )
    still = ET.parse(ROOT / "three-stills.svg").getroot()
    blue = [
        node
        for node in still.findall(".//s:rect", ns)
        if node.attrib.get("fill") == "#4673a8"
    ]
    need(
        len(blue) == 3 and len({node.attrib["x"] for node in blue}) == 3,
        "Three stills must visibly draw three distinct square positions",
    )
    routes = (ROOT / "two-possible-routes.svg").read_text(encoding="utf-8")
    need(
        "stroke-dasharray" in routes
        and 'd="M36 147 H102 H175"' in routes
        and 'd="M36 147 C50 95' in routes,
        "Solid and bent dashed route geometry missing",
    )
    safe = (ROOT / "safe-real-view.svg").read_text(encoding="utf-8")
    need(
        'd="M110 129 L117 121' in safe and "SCHEMATIC ONLY" in safe,
        "Adult hand illustration or representation warning missing",
    )
    raw = ET.parse(ROOT / "raw-attempts.svg").getroot()
    need(
        sum(
            node.text == "VALID / INTERRUPTED / NOT RUN"
            for node in raw.findall(".//s:text", ns)
        )
        == 2,
        "Both raw attempt status lanes required",
    )


def links() -> int:
    count = 0
    for path in ROOT.glob("*.md"):
        for url in re.findall(
            r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")
        ):
            if url.startswith(("https://", "http://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            need(target.exists(), f"Broken local link: {path.name} -> {url}")
            count += 1
    return count


def markdown_list_spacing() -> int:
    count = 0
    marker = re.compile(r"^(?:- |\d+\. )")
    for path in ROOT.glob("*.md"):
        lines = path.read_text(encoding="utf-8").splitlines()
        for index, entry in enumerate(lines):
            if not marker.match(entry):
                continue
            count += 1
            previous = lines[index - 1] if index else ""
            following = lines[index + 1] if index + 1 < len(lines) else ""
            need(
                not previous.strip() or bool(marker.match(previous)),
                f"List needs blank line above: {path.name}:{index + 1}",
            )
            need(
                not following.strip() or bool(marker.match(following)),
                f"List needs blank line below: {path.name}:{index + 1}",
            )
    return count


def manifest_data() -> dict:
    files = sorted(
        path
        for path in ROOT.rglob("*")
        if path.is_file()
        and path.name != "manifest.json"
        and "__pycache__" not in path.parts
        and ".ruff_cache" not in path.parts
        and not path.name.startswith("_preview")
    )
    return {
        "pack_id": "subjectnest-acara-v9-foundation-science-t3-w25-26",
        "pack_version": "0.1.0-draft",
        "created_at": "2026-09-30",
        "review_status": "pending_human_educator_accessibility_local_syllabus_safety_privacy_and_classroom_review",
        "year_label": "Australian Curriculum Foundation Year / Queensland Prep",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(CODES),
        "curriculum_source": {
            "authority": "Australian Curriculum, Assessment and Reporting Authority",
            "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx",
            "retrieved_at": "2026-09-29",
            "source_hash": SOURCE_SHA,
            "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use",
        },
        "rights": "Original SubjectNest texts, models and vector aids CC BY 4.0; local Python utilities Apache-2.0; ACARA retains its own terms.",
        "original_assets": [
            {
                "item_code_or_locator": str(path.relative_to(ROOT)),
                "sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
            }
            for path in files
        ],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    rows = official()
    lessons(rows)
    cards_practice_checks()
    assets()
    link_count = links()
    list_count = markdown_list_spacing()
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
        )
    need(
        json.loads(manifest.read_text(encoding="utf-8")) == expected,
        "Hash manifest stale or missing",
    )
    print(
        f"PASS Foundation Science W25–26: 10x25m, 30 routes, 20 swaps, 10 family prompts, 2 fresh checks/keys, 7 exact codes, 6 pictorial A4 aids, {link_count} local links, {list_count} spaced list items, hashes"
    )


if __name__ == "__main__":
    main()
