#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only structural/source/asset verification for Science Weeks 27–28."""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = {
    "AC9SFU03",
    "AC9SFH01",
    "AC9SFI01",
    "AC9SFI02",
    "AC9SFI03",
    "AC9SFI04",
    "AC9SFI05",
}
STEMS = (
    "split-sign",
    "three-swatches",
    "safe-source-gate",
    "two-source-record",
    "one-property-sort",
    "ask-source-explain",
)
TEACHING = (
    "README.md",
    "LESSONS.md",
    "MATERIALS.md",
    "LEARNER-CARDS.md",
    "OPTIONAL-PRACTICE.md",
)


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def official() -> dict[str, dict]:
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    data = json.loads(
        (STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8")
    )
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == SOURCE_SHA
        == data["source_sha256"],
        "Official workbook/import hash differs",
    )
    rows = {
        r["code"]: r
        for r in data["records"]
        if r.get("record_type") == "content_description" and r.get("code") in CODES
    }
    table: dict[str, tuple[int, str]] = {}
    for line in (
        (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8").splitlines()
    ):
        match = re.match(
            r"^\| (AC9[A-Z0-9]+) \| (\d+) \| Science · Foundation Year \| (.*?) \|$",
            line,
        )
        if match:
            table[match.group(1)] = (int(match.group(2)), match.group(3))
    need(set(table) == CODES == set(rows), "Crosswalk/source code mismatch")
    for code, (source_row, description) in table.items():
        row = rows[code]
        need(
            row["attributes"]["level"] == "Foundation Year"
            and row["attributes"]["learning_area"] == "Science",
            f"Wrong level/area: {code}",
        )
        need(
            source_row == row["source_row"]
            and " ".join(description.split()) == " ".join(row["plain_text"].split()),
            f"Wrong source row/text: {code}",
        )
    return rows


def lessons(rows: dict[str, dict]) -> None:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", content, re.MULTILINE))
    need(
        [int(m.group(2)) for m in marks] == list(range(131, 141)),
        "Ten Science days 131–140 required",
    )
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (day - 1) // 5 + 1, f"Wrong week for Day {day}")
        body = content[
            mark.end() : marks[index + 1].start()
            if index + 1 < len(marks)
            else len(content)
        ]
        times = [
            int(value)
            for value in re.findall(
                r"^\d+\. \*\*[^\n]*?\b(\d+) min\.", body, re.MULTILINE
            )
        ]
        need(
            len(times) == 6 and sum(times) == 25,
            f"Day {day} needs six 25-minute steps: {times}",
        )
        need(
            "**Goal:**" in body and "**Prepare:**" in body,
            f"Day {day} goal/preparation missing",
        )
        for code in set(re.findall(r"\bAC9[A-Z0-9]+\b", body)):
            need(code in rows, f"Unknown/non-Foundation lesson code: {code}")
    need(
        "first independent" in content.lower()
        and "REAL OBSERVATION NOT AVAILABLE" in content,
        "Assessment/source boundary absent",
    )


def cards_practice_checks() -> None:
    cards = (ROOT / "LEARNER-CARDS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need(
        [int(m.group(1)) for m in marks] == list(range(131, 141)),
        "Ten learner-card days required",
    )
    for index, mark in enumerate(marks):
        body = cards[
            mark.end() : marks[index + 1].start()
            if index + 1 < len(marks)
            else len(cards)
        ]
        need(
            len(re.findall(r"^- \*\*[^*]+:\*\* .+$", body, re.MULTILINE)) == 3,
            f"Day {mark.group(1)} needs three same-goal routes",
        )
    need(
        "ASSESSMENT.md" not in cards and "TEACHER-KEY.md" not in cards,
        "Held content linked from learner page",
    )

    practice = (ROOT / "OPTIONAL-PRACTICE.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", practice, re.MULTILINE))
    need(
        [int(m.group(1)) for m in marks] == list(range(131, 141)),
        "Ten worked-swap days required",
    )
    for index, mark in enumerate(marks):
        body = practice[
            mark.end() : marks[index + 1].start()
            if index + 1 < len(marks)
            else len(practice)
        ]
        need(
            len(
                re.findall(
                    r"^[12]\. \*\*.+?\*\* .+?\*\*Worked report:\*\* .+$",
                    body,
                    re.MULTILINE,
                )
            )
            == 2,
            f"Day {mark.group(1)} needs two worked swaps",
        )

    assessment = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    teaching = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in TEACHING)
    for phrase in (
        "TRIANGLE PANEL 64",
        "FOUR LARGE DOTS",
        "PATCH NOTE 91",
        "TWO WIDE DOTS",
    ):
        need(
            phrase.lower() not in teaching.lower(),
            f"Held case leaked into teaching: {phrase}",
        )
        need(phrase.lower() in assessment.lower(), f"Held case missing: {phrase}")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    lesson_cases = re.findall(
        r"\[teacher-held Case ([A-Z]+)\]\(ASSESSMENT\.md\)",
        (ROOT / "LESSONS.md").read_text(encoding="utf-8"),
    )
    assessment_cases = re.findall(
        r"^## Day \d+ · fresh Case ([A-Z]+)", assessment, re.MULTILINE
    )
    key_cases = re.findall(r"^## Day \d+ · Case ([A-Z]+)", key, re.MULTILINE)
    need(
        lesson_cases == assessment_cases == key_cases == ["AD", "AE"],
        "Held case IDs differ between lessons, cases and key",
    )
    need(
        all(f"Day {day}" in assessment and f"Day {day}" in key for day in (135, 140)),
        "Two held checks/separate keys required",
    )
    need(
        "first independent" in assessment.lower()
        and "REAL OBSERVATION NOT AVAILABLE" in key
        and "no fixed local material answer" in key.lower()
        and "not the child's own direct sensory observation" in key.lower(),
        "Fresh check or real-source boundary absent",
    )
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8")
    need(
        all(
            term in materials
            for term in (
                "local date",
                "approved",
                "REAL OBSERVATION NOT AVAILABLE",
                "ADULT REPORT",
                "tactile",
                "FICTIONAL MODEL",
                "SOURCE",
            )
        ),
        "Core source/access/safety facts missing",
    )
    family = (ROOT / "FAMILY-OPTIONAL.md").read_text(encoding="utf-8")
    need(
        [int(d) for d in re.findall(r"^\| (\d{2,3}) \|", family, re.MULTILINE)]
        == list(range(131, 141)),
        "Ten optional family prompts required",
    )
    need(
        "no family needs to buy" in family.lower() and "not homework" in family.lower(),
        "Family boundary absent",
    )
    read_aloud = (ROOT / "READ-ALOUD.md").read_text(encoding="utf-8")
    need(
        len(re.findall(r"^## Track [123] ·", read_aloud, re.MULTILINE)) == 3
        and "not recorded audio" in read_aloud.lower()
        and "fictional paper picture" in read_aloud.lower()
        and "invented maker note" in read_aloud.lower(),
        "Three source-labelled narration text masters or audio status missing",
    )


def assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    labels = {
        "split-sign": (
            "PLAIN-LOOKING",
            "SIX DRAWN LINES",
            "MATERIAL NOT VERIFIED",
        ),
        "three-swatches": (
            "SAMPLE A",
            "SAMPLE B",
            "SAMPLE C",
            "MATERIAL WORDS FROM MAKER NOTE",
        ),
        "safe-source-gate": (
            "APPROVAL + LOCAL DATE",
            "CHILD ACCESS CHOICE",
            "ACTUAL FEATURE OR NOT OBSERVED",
        ),
        "two-source-record": (
            "DIRECT FEATURE / ADULT REPORT / MODEL",
            "MATERIAL LABEL + HOW VERIFIED",
            "HIDDEN JOIN UNKNOWN",
        ),
        "one-property-sort": (
            "SOURCE + DATE / FICTIONAL MODEL",
            "VISIBLE RIBS",
            "NO VISIBLE RIBS",
            "UNSORTED",
            "COUNTEREXAMPLE",
        ),
        "ask-source-explain": (
            "ASK A NEUTRAL",
            "SOURCE SAYS",
            "MAKER/PRODUCT SOURCE + WORDS",
            "PRIVATE LISTENER RESPONSE OR NONE",
        ),
    }
    for stem in STEMS:
        svg, pdf = ROOT / f"{stem}.svg", ROOT / f"{stem}.pdf"
        tree = ET.parse(svg).getroot()
        need(
            tree.attrib.get("width") == "210mm"
            and tree.attrib.get("height") == "297mm",
            f"{stem} SVG is not A4",
        )
        need(
            tree.find("s:title", ns) is not None
            and tree.find("s:desc", ns) is not None,
            f"{stem} lacks SVG title/desc",
        )
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        need(
            "Pages:           1" in info and "(A4)" in info,
            f"{stem} PDF is not single A4",
        )
        extracted = " ".join(
            subprocess.check_output(["pdftotext", str(pdf), "-"], text=True).split()
        )
        need(
            all(label in extracted for label in labels[stem])
            and "CC BY 4.0" in extracted,
            f"{stem} PDF text/credit missing",
        )
    split = ET.parse(ROOT / "split-sign.svg").getroot()
    gold_lines = [
        node
        for node in split.findall(".//s:path", ns)
        if node.attrib.get("stroke") == "#a45d00"
    ]
    need(
        len(gold_lines) == 6,
        "Split sign must visibly draw six distinct parallel lines",
    )
    swatches = ET.parse(ROOT / "three-swatches.svg").getroot()
    swatch_paths = swatches.findall(".//s:path", ns)
    need(
        sum(node.attrib.get("stroke") == "#4673a8" for node in swatch_paths) == 4
        and sum(node.attrib.get("stroke") == "#a45d00" for node in swatch_paths) == 4,
        "B/C swatches need distinct real pictorial line geometry",
    )
    safe = (ROOT / "safe-source-gate.svg").read_text(encoding="utf-8")
    need(
        "SCHEMATIC ONLY" in safe
        and "BROAD SECURED SAMPLES" in safe
        and "CHILD ACCESS CHOICE" in safe,
        "Safe source picture/access warning missing",
    )
    record = ET.parse(ROOT / "two-source-record.svg").getroot()
    need(
        sum(
            node.text == "DIRECT FEATURE / ADULT REPORT / MODEL"
            for node in record.findall(".//s:text", ns)
        )
        == 2,
        "Two distinct source/access records required",
    )
    sort = ET.parse(ROOT / "one-property-sort.svg").getroot()
    unsorted_squares = [
        node
        for node in sort.findall(".//s:rect", ns)
        if node.attrib.get("stroke") == "#4673a8"
    ]
    need(
        len(unsorted_squares) == 3
        and {node.attrib.get("y") for node in unsorted_squares} == {"186"},
        "Three empty items must stay in the distinct unsorted tray",
    )
    explain = (ROOT / "ask-source-explain.svg").read_text(encoding="utf-8")
    need(
        'd="M20 59 H94' in explain and 'd="M118 59 H189' in explain,
        "Question and source bubbles missing",
    )


def links() -> int:
    count = 0
    for path in ROOT.glob("*.md"):
        for url in re.findall(
            r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")
        ):
            if url.startswith(("https://", "http://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            need(target.exists(), f"Broken local link: {path.name} -> {url}")
            count += 1
    return count


def markdown_list_spacing() -> int:
    count = 0
    marker = re.compile(r"^(?:- |\d+\. )")
    for path in ROOT.glob("*.md"):
        lines = path.read_text(encoding="utf-8").splitlines()
        for index, entry in enumerate(lines):
            if not marker.match(entry):
                continue
            count += 1
            previous = lines[index - 1] if index else ""
            following = lines[index + 1] if index + 1 < len(lines) else ""
            need(
                not previous.strip() or bool(marker.match(previous)),
                f"List needs blank line above: {path.name}:{index + 1}",
            )
            need(
                not following.strip() or bool(marker.match(following)),
                f"List needs blank line below: {path.name}:{index + 1}",
            )
    return count


def manifest_data() -> dict:
    files = sorted(
        path
        for path in ROOT.rglob("*")
        if path.is_file()
        and path.name != "manifest.json"
        and "__pycache__" not in path.parts
        and ".ruff_cache" not in path.parts
        and not path.name.startswith("_preview")
    )
    return {
        "pack_id": "subjectnest-acara-v9-foundation-science-t3-w27-28",
        "pack_version": "0.1.0-draft",
        "created_at": "2026-09-30",
        "review_status": "pending_human_educator_accessibility_local_syllabus_safety_privacy_and_classroom_review",
        "year_label": "Australian Curriculum Foundation Year / Queensland Prep",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(CODES),
        "curriculum_source": {
            "authority": "Australian Curriculum, Assessment and Reporting Authority",
            "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx",
            "retrieved_at": "2026-09-29",
            "source_hash": SOURCE_SHA,
            "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use",
        },
        "rights": "Original SubjectNest texts, models and vector aids CC BY 4.0; local Python utilities Apache-2.0; ACARA retains its own terms.",
        "original_assets": [
            {
                "item_code_or_locator": str(path.relative_to(ROOT)),
                "sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
            }
            for path in files
        ],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    rows = official()
    lessons(rows)
    cards_practice_checks()
    assets()
    link_count = links()
    list_count = markdown_list_spacing()
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
        )
    need(
        json.loads(manifest.read_text(encoding="utf-8")) == expected,
        "Hash manifest stale or missing",
    )
    print(
        f"PASS Foundation Science W27–28: 10x25m, 30 routes, 20 swaps, 10 family prompts, 2 fresh checks/keys, 7 exact codes, 6 pictorial A4 aids, {link_count} local links, {list_count} spaced list items, hashes"
    )


if __name__ == "__main__":
    main()
