#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Verify authored Foundation English Weeks 35–36 and pinned ACARA source rows."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
ENGLISH = ROOT.parents[1]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = {"AC9EFLA02", "AC9EFLA05", "AC9EFLA06", "AC9EFLA07", "AC9EFLA09", "AC9EFLE02", "AC9EFLY01", "AC9EFLY02", "AC9EFLY05", "AC9EFLY06", "AC9EFLY07", "AC9EFLY10", "AC9EFLY12", "AC9EFLY13"}
PRINT = {"a", "an", "at", "can", "cap", "cup", "i", "in", "is", "it", "map", "mat", "on", "red", "see", "sit", "tap", "ten", "tin"}
HELD = {"ban", "dab", "bid", "pop"}
TEACHING = ("README.md", "LESSONS.md", "MATERIALS.md", "LEARNER-CARDS.md", "OPTIONAL-PRACTICE.md")


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def official() -> dict[str, dict]:
    workbook = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    data = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8"))
    need(hashlib.sha256(workbook.read_bytes()).hexdigest() == SOURCE_SHA == data["source_sha256"], "Official workbook/import hash differs")
    rows = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description" and r.get("code") in CODES}
    table: dict[str, tuple[int, str]] = {}
    for line in (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8").splitlines():
        match = re.match(r"^\| (AC9[A-Z0-9]+) \| (\d+) \| English · Foundation Year \| (.*?) \|$", line)
        if match:
            table[match.group(1)] = (int(match.group(2)), match.group(3))
    need(set(table) == CODES == set(rows), f"Crosswalk/source code mismatch: {set(table) ^ CODES}")
    for code, (source_row, description) in table.items():
        row = rows[code]
        need(row["attributes"]["level"] == "Foundation Year" and row["attributes"]["learning_area"] == "English", f"Wrong level/area: {code}")
        need(source_row == row["source_row"] and " ".join(description.split()) == " ".join(row["plain_text"].split()), f"Wrong source row/text: {code}")
    return rows


def lessons(rows: dict[str, dict]) -> None:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", content, re.MULTILINE))
    need([int(m.group(2)) for m in marks] == list(range(171, 181)), "Ten English days 171–180 required")
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (day - 1) // 5 + 1, f"Wrong week for Day {day}")
        body = content[mark.end(): marks[index + 1].start() if index + 1 < len(marks) else len(content)]
        times = [int(value) for value in re.findall(r"^\d+\. \*\*[^\n]*?\b(\d+) min\.", body, re.MULTILINE)]
        need(len(times) == 6 and sum(times) == 25, f"Day {day} needs six 25-minute steps, got {times}")
        need("**Goal:**" in body and "**Prepare:**" in body, f"Day {day} goal/preparation format missing")
        for code in set(re.findall(r"\bAC9[A-Z0-9]+\b", body)):
            need(code in rows, f"Unknown/non-Foundation lesson code: {code}")
    need("first independent" in content.lower() and "Adult reads" in content, "Assessment and adult-read boundary absent")


def cards_and_checks() -> None:
    cards = (ROOT / "LEARNER-CARDS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need([int(m.group(1)) for m in marks] == list(range(171, 181)), "Ten learner-card days required")
    for index, mark in enumerate(marks):
        body = cards[mark.end(): marks[index + 1].start() if index + 1 < len(marks) else len(cards)]
        lines = re.findall(r"^- \*\*[^*]+:\*\* .+?\*\*Child print:\*\* “([^”]+)”", body, re.MULTILINE)
        need(len(lines) == 3, f"Day {mark.group(1)} needs three same-target choice routes")
        for line in lines:
            words = set(re.findall(r"[a-z]+", line.lower()))
            need(words <= PRINT, f"Untaught child-print word Day {mark.group(1)}: {words - PRINT}")
    need("ASSESSMENT.md" not in cards and "TEACHER-KEY.md" not in cards, "Teacher-held content linked from learner page")
    teaching = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in TEACHING)
    prior = "\n".join(
        path.read_text(encoding="utf-8")
        for path in ENGLISH.rglob("*.md")
        if ROOT not in path.parents
        and path.name in TEACHING
        and ("term-4" not in path.parts or path.parent.name in {"weeks-31-32", "weeks-33-34"})
    )
    assessment = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    for word in HELD:
        need(re.search(rf"\b{word}\b", teaching, re.IGNORECASE) is None, f"Held-out print item leaked into teaching: {word}")
        need(re.search(rf"\b{word}\b", prior, re.IGNORECASE) is None, f"Held-out print item appeared earlier: {word}")
        need(re.search(rf"\b{word}\b", assessment, re.IGNORECASE) is not None, f"Held-out item missing: {word}")
    for name in ("Nell", "Orin"):
        need(re.search(rf"\b{name}\b", teaching, re.IGNORECASE) is None, f"Fresh check character leaked: {name}")
    need("first independent" in assessment and "not administered" in assessment, "Fresh-check first response/taught-point boundary absent")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    need(all(f"Day {day}" in assessment and f"Day {day}" in key for day in (175, 180)), "Two distinct held-out checks/keys required")
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8")
    need(all(term in materials for term in ("one wide paper mat", "the turns happen", "CIRCLE · CIRCLE · SQUARE", "triangle above the square", "not tested")), "Teaching source facts missing")
    need(all(term in assessment for term in ("four broad paper fossil cards", "one wide paper map", "SQUARE · TRIANGLE · TRIANGLE", "near a square")), "Fresh source facts missing")


def assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    for stem, labels in (
        ("shared-mat-story", ("SCENE 1", "Luma", "Jay", "SCENE 4")),
        ("fair-plan", ("REQUEST", "PROPOSED TRY", "ACTUAL RESULT OR NOT TESTED")),
        ("pattern-shape-cards", ("1 CIRCLE", "2 CIRCLE", "3 SQUARE", "TRIANGLE")),
        ("instruction-strip", ("MATERIALS", "STEP 1", "STEP 2", "HOW TO CHECK")),
        ("listener-repair", ("FIRST WORDS", "WHAT LISTENER DID", "REVISED WORDS", "STILL TO CHECK")),
    ):
        svg, pdf = ROOT / f"{stem}.svg", ROOT / f"{stem}.pdf"
        tree = ET.parse(svg).getroot()
        need(tree.attrib.get("width") == "210mm" and tree.attrib.get("height") == "297mm", f"{stem} SVG is not A4")
        need(tree.find("s:title", ns) is not None and tree.find("s:desc", ns) is not None, f"{stem} SVG lacks title/desc")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        need("Pages:           1" in info and "(A4)" in info, f"{stem} PDF is not single A4")
        extracted = " ".join(subprocess.check_output(["pdftotext", str(pdf), "-"], text=True).split())
        need(all(label in extracted for label in labels) and "CC BY 4.0" in extracted, f"{stem} PDF text/credit missing")
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8").lower()
    need(all(term in materials for term in ("tactile", "request", "not tested", "colour", "instruction")), "Exact nonvisual alternatives absent")


def links() -> int:
    count = 0
    for path in ROOT.glob("*.md"):
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("https://", "http://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            need(target.exists(), f"Broken local link: {path.name} -> {url}")
            count += 1
    return count


def manifest_data() -> dict:
    files = sorted(
        path for path in ROOT.rglob("*")
        if path.is_file() and path.name != "manifest.json" and "__pycache__" not in path.parts and not path.name.startswith("_preview")
    )
    return {
        "pack_id": "subjectnest-acara-v9-foundation-english-t4-w35-36",
        "pack_version": "0.1.0-draft",
        "created_at": "2026-09-29",
        "review_status": "pending_human_educator_accessibility_local_syllabus_safety_privacy_and_classroom_review",
        "year_label": "Australian Curriculum Foundation Year / Queensland Prep",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(CODES),
        "curriculum_source": {
            "authority": "Australian Curriculum, Assessment and Reporting Authority",
            "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx",
            "retrieved_at": "2026-09-29",
            "source_hash": SOURCE_SHA,
            "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use",
        },
        "rights": "Original SubjectNest texts, scripts, checks and vector aids CC BY 4.0; ACARA retains its own terms.",
        "original_assets": [
            {"item_code_or_locator": str(path.relative_to(ROOT)), "sha256": hashlib.sha256(path.read_bytes()).hexdigest()}
            for path in files
        ],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    rows = official()
    lessons(rows)
    cards_and_checks()
    assets()
    link_count = links()
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
    need(json.loads(manifest.read_text(encoding="utf-8")) == expected, "Hash manifest stale or missing")
    print(f"PASS Foundation English W35–36: 10x25m, 30 routes, 2 fresh checks/keys, 14 exact codes, 5 A4 aids, {link_count} local links, hashes")


if __name__ == "__main__":
    main()
