#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only source, structure, printable and hash verification for W29–30."""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = {
    "AC9SFU01",
    "AC9SFH01",
    "AC9SFI01",
    "AC9SFI02",
    "AC9SFI03",
    "AC9SFI04",
    "AC9SFI05",
}
STEMS = (
    "four-notebook-figures",
    "pointed-rule-sort",
    "four-source-gate",
    "two-pair-log",
    "revise-rule-board",
    "feature-share-strip",
)
TEACHING = (
    "README.md",
    "LESSONS.md",
    "MATERIALS.md",
    "LEARNER-CARDS.md",
    "OPTIONAL-PRACTICE.md",
    "READ-ALOUD.md",
    "FAMILY-OPTIONAL.md",
)


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def official_rows() -> dict[str, dict]:
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    data = json.loads(
        (STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8")
    )
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == SOURCE_SHA
        == data["source_sha256"],
        "Official workbook and import hash differ",
    )
    rows = {
        row["code"]: row
        for row in data["records"]
        if row.get("record_type") == "content_description" and row.get("code") in CODES
    }
    table: dict[str, tuple[int, str]] = {}
    for line in (
        (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8").splitlines()
    ):
        match = re.match(
            r"^\| (AC9[A-Z0-9]+) \| (\d+) \| Science · Foundation Year \| (.*?) \|$",
            line,
        )
        if match:
            table[match.group(1)] = (int(match.group(2)), match.group(3))
    need(set(table) == CODES == set(rows), "Crosswalk/source code mismatch")
    for code, (source_row, wording) in table.items():
        row = rows[code]
        need(
            row["attributes"]["learning_area"] == "Science"
            and row["attributes"]["level"] == "Foundation Year"
            and source_row == row["source_row"]
            and " ".join(wording.split()) == " ".join(row["plain_text"].split()),
            f"Wrong source row, level, area or wording: {code}",
        )
    return rows


def lessons(rows: dict[str, dict]) -> None:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", content, re.MULTILINE))
    need(
        [int(mark.group(2)) for mark in marks] == list(range(141, 151)),
        "Ten daily scripts 141–150 required",
    )
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (day - 1) // 5 + 1, f"Wrong week for Day {day}")
        body = content[
            mark.end() : marks[index + 1].start()
            if index + 1 < len(marks)
            else len(content)
        ]
        minutes = [
            int(value)
            for value in re.findall(
                r"^\d+\. \*\*[^\n]*?\b(\d+) min\.", body, re.MULTILINE
            )
        ]
        need(
            len(minutes) == 6 and sum(minutes) == 25,
            f"Day {day} needs six steps totaling 25 minutes: {minutes}",
        )
        need(
            "**Goal:**" in body and "**Prepare:**" in body,
            f"Day {day} lacks goal or preparation",
        )
        for code in re.findall(r"\bAC9[A-Z0-9]+\b", body):
            need(code in rows, f"Non-Foundation code on Day {day}: {code}")
    need(
        "first independent" in content.lower()
        and "REAL OBSERVATION NOT AVAILABLE" in content,
        "Source and assessment boundaries missing",
    )


def teaching_and_checks() -> None:
    cards = (ROOT / "LEARNER-CARDS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need(
        [int(mark.group(1)) for mark in marks] == list(range(141, 151)),
        "Ten learner-card days required",
    )
    for index, mark in enumerate(marks):
        body = cards[
            mark.end() : marks[index + 1].start()
            if index + 1 < len(marks)
            else len(cards)
        ]
        need(
            len(re.findall(r"^- \*\*[^*]+:\*\* .+$", body, re.MULTILINE)) == 3,
            f"Day {mark.group(1)} needs three same-target routes",
        )
    need(
        "ASSESSMENT.md" not in cards and "TEACHER-KEY.md" not in cards,
        "Held cases leaked into learner card links",
    )

    practice = (ROOT / "OPTIONAL-PRACTICE.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", practice, re.MULTILINE))
    need(
        [int(mark.group(1)) for mark in marks] == list(range(141, 151)),
        "Ten optional practice days required",
    )
    for index, mark in enumerate(marks):
        body = practice[
            mark.end() : marks[index + 1].start()
            if index + 1 < len(marks)
            else len(practice)
        ]
        need(
            len(
                re.findall(
                    r"^[12]\. \*\*.+?\*\* .+?\*\*Worked report:\*\* .+$",
                    body,
                    re.MULTILINE,
                )
            )
            == 2,
            f"Day {mark.group(1)} needs two worked interest swaps",
        )

    assessment = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    teaching = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in TEACHING)
    for phrase in (
        "SMOOTH WIDE BORDER 72",
        "THREE-TOOTH ZIGZAG 83",
        "FOUR LARGE SPOTS 96",
        "NOT VISIBLE UNDER FLAP 54",
    ):
        need(phrase in assessment, f"Fresh case detail absent: {phrase}")
        need(phrase not in teaching, f"Held case detail leaked: {phrase}")
    lesson_ids = re.findall(
        r"\[teacher-held Case ([A-Z]+)\]\(ASSESSMENT\.md\)",
        (ROOT / "LESSONS.md").read_text(encoding="utf-8"),
    )
    case_ids = re.findall(
        r"^## Day \d+ · fresh Case ([A-Z]+)", assessment, re.MULTILINE
    )
    key_ids = re.findall(r"^## Day \d+ · Case ([A-Z]+)", key, re.MULTILINE)
    need(
        lesson_ids == case_ids == key_ids == ["AF", "AG"],
        "Held case IDs differ between lessons, assessment and key",
    )
    need(
        all(f"Day {day}" in assessment and f"Day {day}" in key for day in (145, 150))
        and "first independent" in assessment.lower()
        and "REAL OBSERVATION NOT AVAILABLE" in key
        and "not the child's own direct sensory observation" in key,
        "Two fresh checks, keys or source boundary missing",
    )
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8")
    need(
        all(
            term in materials
            for term in (
                "four distinct subject IDs",
                "local date",
                "approved",
                "REAL OBSERVATION NOT AVAILABLE",
                "ADULT REPORT",
                "FICTIONAL MODEL",
                "tactile",
            )
        ),
        "Source, safety or exact access route missing",
    )
    family = (ROOT / "FAMILY-OPTIONAL.md").read_text(encoding="utf-8")
    need(
        [int(day) for day in re.findall(r"^\| (\d{3}) \|", family, re.MULTILINE)]
        == list(range(141, 151))
        and "not homework" in family.lower()
        and "no family needs to buy" in family.lower(),
        "Ten no-burden family options missing",
    )
    read_aloud = (ROOT / "READ-ALOUD.md").read_text(encoding="utf-8")
    need(
        len(re.findall(r"^## Track [123] ·", read_aloud, re.MULTILINE)) == 3
        and "not recorded audio" in read_aloud.lower()
        and "fictional paper notebook" in read_aloud.lower(),
        "Three source-labelled audio text masters missing",
    )


def assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    labels = {
        "four-notebook-figures": (
            "N1 · PLANT DRAWING",
            "N2 · PLANT DRAWING",
            "N3 · ANIMAL PUPPET",
            "N4 · ANIMAL PUPPET",
            "DRAWN MARKS, NOT REAL ORGANISMS",
        ),
        "pointed-rule-sort": (
            "UNSORTED",
            "POINTED OUTER MARK",
            "ROUNDED OUTER MARK",
            "COUNTEREXAMPLE + SOURCE",
        ),
        "four-source-gate": (
            "P1 · PLANT 1",
            "P2 · PLANT 2",
            "A1 · ANIMAL 1",
            "A2 · ANIMAL 2",
            "STOP OR MODEL ONLY",
        ),
        "two-pair-log": (
            "PLANT PAIR",
            "ANIMAL PAIR",
            "VISIBLE / NOT VISIBLE",
            "UNCLEAR · SOURCE + DATE",
        ),
        "revise-rule-board": (
            "FIRST RULE",
            "COUNTEREXAMPLE + SOURCE",
            "SMALLER RULE / QUESTION",
        ),
        "feature-share-strip": (
            "SOURCE TYPE + DATE",
            "COUNTEREXAMPLE",
            "LISTENER RESPONSE OR NONE",
        ),
    }
    for stem in STEMS:
        svg = ROOT / f"{stem}.svg"
        pdf = ROOT / f"{stem}.pdf"
        root = ET.parse(svg).getroot()
        need(
            root.attrib.get("width") == "210mm"
            and root.attrib.get("height") == "297mm"
            and root.find("s:title", ns) is not None
            and root.find("s:desc", ns) is not None,
            f"{stem} lacks A4/accessible SVG metadata",
        )
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        need(
            "Pages:           1" in info and "(A4)" in info,
            f"{stem} PDF is not one A4 page",
        )
        extracted = " ".join(
            subprocess.check_output(["pdftotext", str(pdf), "-"], text=True).split()
        )
        need(
            all(word in extracted for word in labels[stem])
            and "CC BY 4.0" in extracted,
            f"{stem} selectable text or rights credit missing",
        )
    figures = ET.parse(ROOT / "four-notebook-figures.svg").getroot()
    need(
        len(
            [
                item
                for item in figures.findall(".//s:ellipse", ns)
                if item.attrib.get("stroke") == "#377452"
            ]
        )
        == 3
        and len(
            [
                item
                for item in figures.findall(".//s:path", ns)
                if item.attrib.get("fill") == "#e9f3f0"
                and item.attrib.get("stroke") == "#377452"
            ]
        )
        == 3,
        "Plant pictures need three rounded and three pointed leaves",
    )
    gate = (ROOT / "four-source-gate.svg").read_text(encoding="utf-8")
    need(
        all(f"{name} ·" in gate for name in ("P1", "P2", "A1", "A2"))
        and "STOP OR MODEL ONLY" in gate,
        "Four genuine source slots or stop gate missing",
    )


def links() -> int:
    count = 0
    for path in ROOT.glob("*.md"):
        for url in re.findall(
            r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")
        ):
            if url.startswith(("https://", "http://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            need(target.exists(), f"Broken local link: {path.name} -> {url}")
            count += 1
    return count


def markdown_spacing() -> int:
    marker = re.compile(r"^(?:- |\d+\. )")
    count = 0
    for path in ROOT.glob("*.md"):
        lines = path.read_text(encoding="utf-8").splitlines()
        for index, entry in enumerate(lines):
            if not marker.match(entry):
                continue
            count += 1
            before = lines[index - 1] if index else ""
            after = lines[index + 1] if index + 1 < len(lines) else ""
            need(
                not before.strip() or bool(marker.match(before)),
                f"List needs blank line above: {path.name}:{index + 1}",
            )
            need(
                not after.strip() or bool(marker.match(after)),
                f"List needs blank line below: {path.name}:{index + 1}",
            )
    return count


def manifest_data() -> dict:
    files = sorted(
        path
        for path in ROOT.rglob("*")
        if path.is_file()
        and path.name != "manifest.json"
        and "__pycache__" not in path.parts
        and ".ruff_cache" not in path.parts
        and not path.name.startswith("_preview")
    )
    return {
        "pack_id": "subjectnest-acara-v9-foundation-science-t3-w29-30",
        "pack_version": "0.1.0-draft",
        "created_at": "2026-09-30",
        "review_status": "pending_human_educator_accessibility_local_syllabus_safety_privacy_and_classroom_review",
        "year_label": "Australian Curriculum Foundation Year / Queensland Prep",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(CODES),
        "curriculum_source": {
            "authority": "Australian Curriculum, Assessment and Reporting Authority",
            "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx",
            "retrieved_at": "2026-09-29",
            "source_hash": SOURCE_SHA,
            "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use",
        },
        "rights": "Original SubjectNest texts, models and vector aids CC BY 4.0; local Python utilities Apache-2.0; ACARA and QCAA retain their own terms.",
        "third_party_notice": "FONT-LICENSE.txt reproduces the DejaVu/Bitstream font notice for PDF-embedded subsets; ACARA curriculum wording remains under its own terms.",
        "pack_files": [
            {
                "item_code_or_locator": str(path.relative_to(ROOT)),
                "sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
            }
            for path in files
        ],
        "original_assets": [
            {
                "item_code_or_locator": str(path.relative_to(ROOT)),
                "sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
            }
            for path in files
            if path.name not in {"FONT-LICENSE.txt", "CURRICULUM-CROSSWALK.md"}
        ],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    rows = official_rows()
    lessons(rows)
    teaching_and_checks()
    assets()
    link_count = links()
    list_count = markdown_spacing()
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
        )
    need(
        json.loads(manifest.read_text(encoding="utf-8")) == expected,
        "Hash manifest missing or stale",
    )
    print(
        f"PASS Foundation Science W29–30: 10x25m, 30 routes, 20 swaps, 10 family prompts, 2 fresh checks/keys, 7 exact codes, 6 pictorial A4 aids, {link_count} local links, {list_count} spaced list items, hashes"
    )


if __name__ == "__main__":
    main()
