#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only source, lesson, print, interactive and hash verification."""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import tempfile
import warnings
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

import openpyxl
from PIL import ImageFont

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = {
    "AC9SFU03",
    "AC9SFH01",
    "AC9SFI01",
    "AC9SFI02",
    "AC9SFI03",
    "AC9SFI04",
    "AC9SFI05",
}
STEMS = (
    "puppet-sample-mat",
    "view-setup",
    "prediction-view-log",
    "layer-pair",
    "window-parts",
    "helper-share",
)
TEACHING = (
    "README.md",
    "LESSONS.md",
    "MATERIALS.md",
    "LEARNER-CARDS.md",
    "OPTIONAL-PRACTICE.md",
    "READ-ALOUD.md",
    "FAMILY-OPTIONAL.md",
)


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def sources() -> None:
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    data = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text())
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == SOURCE_SHA
        == data["source_sha256"],
        "Official source hash mismatch",
    )
    rows = {
        r["code"]: r
        for r in data["records"]
        if r.get("record_type") == "content_description" and r.get("code") in CODES
    }
    table = {}
    for line in (ROOT / "CURRICULUM-CROSSWALK.md").read_text().splitlines():
        m = re.match(
            r"^\| (AC9[A-Z0-9]+) \| (\d+) \| Science · Foundation Year \| (.*?) \|$",
            line,
        )
        if m:
            table[m[1]] = (int(m[2]), m[3])
    need(set(rows) == set(table) == CODES, "Wrong crosswalk codes")
    with warnings.catch_warnings():
        warnings.simplefilter("ignore", UserWarning)
        source = openpyxl.load_workbook(workbook, read_only=True, data_only=True)
    sheet = source["Learning areas"]
    wanted = {n for n, _ in table.values()}
    direct = {}
    for n, values in enumerate(
        sheet.iter_rows(min_row=min(wanted), max_row=max(wanted), values_only=True),
        min(wanted),
    ):
        if n in wanted:
            direct[n] = values
    source.close()
    for code, (n, wording) in table.items():
        row = rows[code]
        values = direct[n]
        need(
            row["source_row"] == n
            and row["attributes"]["learning_area"] == "Science"
            and row["attributes"]["level"] == "Foundation Year",
            f"Wrong source level/row: {code}",
        )
        need(
            values[0] == "Science"
            and values[2] == "Foundation Year"
            and values[4] == code
            and values[9] == wording == row["plain_text"],
            f"Direct workbook/import/wording mismatch: {code}",
        )


def lessons_and_checks() -> None:
    lesson = (ROOT / "LESSONS.md").read_text()
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", lesson, re.MULTILINE))
    need(
        [int(m[2]) for m in marks] == list(range(151, 161)),
        "Ten daily scripts 151–160 required",
    )
    for i, m in enumerate(marks):
        day = int(m[2])
        body = lesson[
            m.end() : marks[i + 1].start() if i + 1 < len(marks) else len(lesson)
        ]
        minutes = [
            int(v)
            for v in re.findall(
                r"^\d+\. \*\*[^\n]*?\b(\d+) min\.\*\*", body, re.MULTILINE
            )
        ]
        need(
            int(m[1]) == (day - 1) // 5 + 1
            and len(minutes) == 6
            and sum(minutes) == 25,
            f"Day {day} timing/week wrong",
        )
        need(
            "**Goal:**" in body and "**Prepare:**" in body,
            f"Day {day} preparation/goal absent",
        )
        need(
            set(re.findall(r"\bAC9[A-Z0-9]+\b", body)) <= CODES,
            f"Day {day} code outside source scope",
        )
    for name, count, pattern in (
        ("LEARNER-CARDS.md", 3, r"^- \*\*.+?:\*\* .+$"),
        (
            "OPTIONAL-PRACTICE.md",
            2,
            r"^[12]\. \*\*.+?\*\* .+?\*\*Worked report:\*\* .+$",
        ),
    ):
        content = (ROOT / name).read_text()
        blocks = list(re.finditer(r"^## Day (\d+)\b", content, re.MULTILINE))
        need(
            [int(m[1]) for m in blocks] == list(range(151, 161)),
            f"{name}: days missing",
        )
        for i, m in enumerate(blocks):
            body = content[
                m.end() : blocks[i + 1].start() if i + 1 < len(blocks) else len(content)
            ]
            need(
                len(re.findall(pattern, body, re.MULTILINE)) == count,
                f"{name} Day{m[1]} routes/swaps absent",
            )
    assessment = (ROOT / "ASSESSMENT.md").read_text()
    key = (ROOT / "TEACHER-KEY.md").read_text()
    teaching = "\n".join((ROOT / name).read_text() for name in TEACHING)
    for token in (
        "KITE WINDOW 61",
        "POSTCARD VIEWER 84",
        "grey card, ID GC",
        "clear panel, ID CP",
    ):
        need(
            token in assessment and token not in teaching,
            f"Fresh case absent or leaked: {token}",
        )
    need(
        re.findall(r"teacher-held Case ([A-Z]+)", lesson)
        == re.findall(r"^## Day \d+ · fresh Case ([A-Z]+)", assessment, re.MULTILINE)
        == re.findall(r"^## Day \d+ · Case ([A-Z]+)", key, re.MULTILINE)
        == ["AH", "AI"],
        "Fresh case/key IDs differ",
    )
    need(
        "not the child's own direct sensory observation" in key
        and "REAL OBSERVATION NOT AVAILABLE" in key
        and "first independent" in assessment,
        "Assessment evidence boundary absent",
    )
    material = (ROOT / "MATERIALS.md").read_text()
    need(
        all(
            x in material
            for x in (
                "ACTUAL VIEW",
                "ADULT REPORT",
                "FICTIONAL MODEL",
                "REAL OBSERVATION NOT AVAILABLE",
                "local date",
                "tactile",
            "material identity not confirmed",
            "Do not say the cover hid the circle",
            )
        ),
        "Actual/access/source boundary absent",
    )
    family = (ROOT / "FAMILY-OPTIONAL.md").read_text()
    need(
        [int(n) for n in re.findall(r"^\| (\d{3}) \|", family, re.MULTILINE)]
        == list(range(151, 161))
        and "not homework" in family
        and "No family needs to buy" in family,
        "Family options incomplete",
    )
    aloud = (ROOT / "READ-ALOUD.md").read_text()
    need(
        len(re.findall(r"^## Track [123] ·", aloud, re.MULTILINE)) == 3
        and "not recorded audio" in aloud,
        "Narration status/master count wrong",
    )


def assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    alternatives = (ROOT / "print/TEXT-ALTERNATIVES.md").read_text()
    font_root = Path("/usr/share/fonts/truetype/dejavu")
    for stem in STEMS:
        path = ROOT / "print" / stem
        root = ET.parse(path.with_suffix(".svg")).getroot()
        need(
            root.attrib.get("width") == "210mm"
            and root.attrib.get("height") == "297mm"
            and root.find("s:title", ns) is not None
            and root.find("s:desc", ns) is not None,
            f"A4 SVG metadata: {stem}",
        )
        info = subprocess.check_output(
            ["pdfinfo", str(path.with_suffix(".pdf"))], text=True
        )
        need("Pages:           1" in info and "(A4)" in info, f"PDF not one A4: {stem}")
        pdfwords = " ".join(
            subprocess.check_output(
                ["pdftotext", str(path.with_suffix(".pdf")), "-"], text=True
            ).split()
        )
        for node in root.findall(".//s:text", ns):
            words = "".join(node.itertext())
            need(
                " ".join(words.split()) in pdfwords and "- " + words in alternatives,
                f"Visible text/alternative mismatch: {stem}: {words}",
            )
            x, y, size = (
                float(node.attrib["x"]),
                float(node.attrib["y"]),
                float(node.attrib["font-size"]),
            )
            font = ImageFont.truetype(
                str(
                    font_root
                    / (
                        "DejaVuSans-Bold.ttf"
                        if node.attrib.get("font-weight") == "700"
                        else "DejaVuSans.ttf"
                    )
                ),
                round(size * 100),
            )
            width = font.getlength(words) / 100
            need(
                10 <= x and x + width <= 200 and size <= y <= 295,
                f"Text exceeds page safe bounds: {stem}: {words}",
            )
    with tempfile.TemporaryDirectory(prefix="subjectnest-science-print-") as tmp:
        subprocess.run(
            ["python", str(ROOT / "print/generate_print.py"), "--output", tmp],
            check=True,
            stdout=subprocess.DEVNULL,
        )
        names = [f"{stem}.{ext}" for stem in STEMS for ext in ("svg", "pdf")] + [
            "TEXT-ALTERNATIVES.md"
        ]
        need(
            all(
                (Path(tmp) / name).read_bytes() == (ROOT / "print" / name).read_bytes()
                for name in names
            ),
            "Print reproduction differs",
        )


def interactive() -> None:
    source = (ROOT / "interactive/window-lab.html").read_text()
    scripts = re.findall(r"<script>([\s\S]*?)</script>", source)
    need(len(scripts) == 1, "Expected one inline script")
    need(
        not re.search(
            r"(?:https?:)?//[^\s]+",
            scripts[0].replace("http://www.w3.org/2000/svg", ""),
        ),
        "External script URL",
    )
    need(
        not re.search(
            r"\b(fetch|XMLHttpRequest|localStorage|sessionStorage|indexedDB|getUserMedia|Audio|WebSocket)\b",
            scripts[0],
        ),
        "Unexpected network/storage/media API",
    )
    with tempfile.TemporaryDirectory(prefix="subjectnest-science-js-") as tmp:
        path = Path(tmp) / "lab.js"
        path.write_text(scripts[0])
        subprocess.run(
            ["node", "--check", str(path)], check=True, stdout=subprocess.DEVNULL
        )
    need(
        all(
            x in source
            for x in (
                'id="cover"',
                'id="pictures" aria-pressed="true"',
                'id="words" aria-pressed="false"',
                'role="status" aria-live="polite"',
                'scope="row"',
                "without JavaScript",
                'cover.value="P";view="Pictures"',
            )
        ),
        "Interactive controls/fallback/reset absent",
    )
    need(
        "Pip's pretend window lab" in source and "invented record" in source,
        "Interactive source label absent",
    )
    for phrase in (
        "I can see a pale circle shape. Its edge looks fuzzy.",
        "I cannot see the circle through the sheet.",
        "I can see the circle and its edge.",
        "I cannot make out the circle through these two sheets.",
    ):
        need(
            source.count(phrase) == (3 if phrase.startswith("I can see a pale") else 2)
            and phrase in (ROOT / "MATERIALS.md").read_text(),
            "Interactive/table/material record disagreement",
        )


def links_and_spacing() -> int:
    count = 0
    for path in ROOT.rglob("*.md"):
        content = path.read_text()
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", content):
            if url.startswith(("http://", "https://", "mailto:", "#")):
                continue
            need(
                (path.parent / unquote(url.partition("#")[0])).resolve().exists(),
                f"Broken local link {path.name}: {url}",
            )
            count += 1
        lines = content.splitlines()
        marker = re.compile(r"^(?:- |\d+\. )")
        for i, line in enumerate(lines):
            if marker.match(line):
                before = lines[i - 1] if i else ""
                after = lines[i + 1] if i + 1 < len(lines) else ""
                need(
                    not before.strip() or bool(marker.match(before)),
                    f"List blank line above: {path.name}:{i + 1}",
                )
                need(
                    not after.strip() or bool(marker.match(after)),
                    f"List blank line below: {path.name}:{i + 1}",
                )
    for url in re.findall(
        r'href="([^"]+)"', (ROOT / "interactive/window-lab.html").read_text()
    ):
        if not url.startswith(("http://", "https://", "#")):
            need(
                (ROOT / "interactive" / unquote(url)).resolve().exists(),
                f"Interactive local link missing: {url}",
            )
            count += 1
    return count


def manifest_data() -> dict:
    files = sorted(
        p
        for p in ROOT.rglob("*")
        if p.is_file()
        and p.name != "manifest.json"
        and not {"__pycache__", ".ruff_cache"}.intersection(p.parts)
    )
    entries = [
        {
            "item_code_or_locator": str(p.relative_to(ROOT)),
            "sha256": hashlib.sha256(p.read_bytes()).hexdigest(),
        }
        for p in files
    ]
    return {
        "pack_id": "subjectnest-acara-v9-foundation-science-t4-w31-32",
        "pack_version": "0.1.0-draft",
        "created_at": "2026-09-30",
        "review_status": "author_and_independent_agent_review_only_pending_human_educator_accessibility_local_syllabus_safety_and_classroom_review",
        "year_label": "Australian Curriculum Foundation Year / Queensland Prep",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(CODES),
        "curriculum_source": {
            "authority": "Australian Curriculum, Assessment and Reporting Authority",
            "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx",
            "retrieved_at": "2026-09-29",
            "source_hash": SOURCE_SHA,
            "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use",
        },
        "rights": "Original SubjectNest lessons, models, drawings and text CC BY 4.0; code Apache-2.0; official source/font terms retained.",
        "pack_files": entries,
        "original_assets": [
            entry
            for entry in entries
            if entry["item_code_or_locator"]
            not in {
                "print/FONT-LICENSE.txt",
                "CODE-LICENSE.txt",
                "CURRICULUM-CROSSWALK.md",
            }
        ],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    sources()
    lessons_and_checks()
    assets()
    interactive()
    count = links_and_spacing()
    data = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(json.dumps(data, ensure_ascii=False, indent=2) + "\n")
    need(json.loads(manifest.read_text()) == data, "Hash manifest absent or stale")
    print(
        f"PASS Science W31–32: 10×25m, 30 routes, 20 worked choices, 10 home choices, 2 fresh checks, 7 direct workbook rows, 6 A4 pairs, 13 reproduced files, offline lab, {count} local links, {len(data['pack_files'])} SHA files; no classroom result claimed"
    )


if __name__ == "__main__":
    main()
