#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Verify authored Foundation English Weeks 39–40 and pinned ACARA source rows."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
ENGLISH = ROOT.parents[1]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = {"AC9EFLA05", "AC9EFLA07", "AC9EFLA09", "AC9EFLE02", "AC9EFLE03", "AC9EFLE05", "AC9EFLY01", "AC9EFLY02", "AC9EFLY03", "AC9EFLY04", "AC9EFLY05", "AC9EFLY06", "AC9EFLY07", "AC9EFLY10", "AC9EFLY12", "AC9EFLY13"}
PRINT = {"a", "an", "at", "can", "cap", "cup", "i", "in", "is", "it", "map", "mat", "on", "red", "see", "sit", "tap", "ten", "tin"}
HELD = {"rag", "dim", "nut", "cob"}
TEACHING = ("README.md", "LESSONS.md", "MATERIALS.md", "LEARNER-CARDS.md", "OPTIONAL-PRACTICE.md")


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def official() -> dict[str, dict]:
    workbook = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    data = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8"))
    need(hashlib.sha256(workbook.read_bytes()).hexdigest() == SOURCE_SHA == data["source_sha256"], "Official workbook/import hash differs")
    rows = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description" and r.get("code") in CODES}
    table: dict[str, tuple[int, str]] = {}
    for line in (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8").splitlines():
        match = re.match(r"^\| (AC9[A-Z0-9]+) \| (\d+) \| English · Foundation Year \| (.*?) \|$", line)
        if match:
            table[match.group(1)] = (int(match.group(2)), match.group(3))
    need(set(table) == CODES == set(rows), f"Crosswalk/source code mismatch: {set(table) ^ CODES}")
    for code, (source_row, description) in table.items():
        row = rows[code]
        need(row["attributes"]["level"] == "Foundation Year" and row["attributes"]["learning_area"] == "English", f"Wrong level/area: {code}")
        need(source_row == row["source_row"] and " ".join(description.split()) == " ".join(row["plain_text"].split()), f"Wrong source row/text: {code}")
    return rows


def lessons(rows: dict[str, dict]) -> None:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", content, re.MULTILINE))
    need([int(m.group(2)) for m in marks] == list(range(191, 201)), "Ten English days 191–200 required")
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (day - 1) // 5 + 1, f"Wrong week for Day {day}")
        body = content[mark.end(): marks[index + 1].start() if index + 1 < len(marks) else len(content)]
        times = [int(value) for value in re.findall(r"^\d+\. \*\*[^\n]*?\b(\d+) min\.", body, re.MULTILINE)]
        need(len(times) == 6 and sum(times) == 25, f"Day {day} needs six 25-minute steps, got {times}")
        need("**Goal:**" in body and "**Prepare:**" in body, f"Day {day} goal/preparation format missing")
        for code in set(re.findall(r"\bAC9[A-Z0-9]+\b", body)):
            need(code in rows, f"Unknown/non-Foundation lesson code: {code}")
    need("first independent" in content.lower() and "Adult reads" in content, "Assessment and adult-read boundary absent")


def cards_and_checks() -> None:
    cards = (ROOT / "LEARNER-CARDS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need([int(m.group(1)) for m in marks] == list(range(191, 201)), "Ten learner-card days required")
    for index, mark in enumerate(marks):
        body = cards[mark.end(): marks[index + 1].start() if index + 1 < len(marks) else len(cards)]
        lines = re.findall(r"^- \*\*[^*]+:\*\* .+?\*\*Child print:\*\* “([^”]+)”", body, re.MULTILINE)
        need(len(lines) == 3, f"Day {mark.group(1)} needs three same-target choice routes")
        for line in lines:
            words = set(re.findall(r"[a-z]+", line.lower()))
            need(words <= PRINT, f"Untaught child-print word Day {mark.group(1)}: {words - PRINT}")
    need("ASSESSMENT.md" not in cards and "TEACHER-KEY.md" not in cards, "Teacher-held content linked from learner page")
    teaching = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in TEACHING)
    prior = "\n".join(
        path.read_text(encoding="utf-8")
        for path in ENGLISH.rglob("*.md")
        if ROOT not in path.parents
        and path.name in TEACHING
        and ("term-4" not in path.parts or path.parent.name in {"weeks-31-32", "weeks-33-34", "weeks-35-36", "weeks-37-38"})
    )
    assessment = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    for word in HELD:
        need(re.search(rf"\b{word}\b", teaching, re.IGNORECASE) is None, f"Held-out print item leaked into teaching: {word}")
        need(re.search(rf"\b{word}\b", prior, re.IGNORECASE) is None, f"Held-out print item appeared earlier: {word}")
        need(re.search(rf"\b{word}\b", assessment, re.IGNORECASE) is not None, f"Held-out item missing: {word}")
    for phrase in ("BAG at the left", "BOOK at the right", "DESK pocket", "Sol clipped", "PRETEND POST DESK"):
        need(phrase.lower() not in teaching.lower(), f"Fresh check source leaked into teaching: {phrase}")
    need("first independent" in assessment.lower() and "not administered" in assessment.lower(), "Fresh-check first response/taught-point boundary absent")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    need(all(f"Day {day}" in assessment and f"Day {day}" in key for day in (195, 200)), "Two distinct held-out checks/keys required")
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8")
    need(all(term in materials for term in ("TENT", "POND", "PATH", "SCENE pocket", "Mira drew a paper map", "STORY CARD STATION", "A map is on a mat.", "NOT AVAILABLE")), "Teaching source facts missing")
    need(all(term in assessment for term in ("BAG", "BOOK", "PENCIL", "DESK pocket", "Sol clipped", "PRETEND POST DESK")), "Fresh source facts missing")


def practice() -> None:
    content = (ROOT / "OPTIONAL-PRACTICE.md").read_text(encoding="utf-8")
    rows = re.findall(r"^\| (19\d|200) \| (.+?) \| (.+?) \| (.+?) \|$", content, re.MULTILINE)
    need([int(row[0]) for row in rows] == list(range(191, 201)), "Optional practice needs two worked swaps on each of ten days")
    for day, first, second, move in rows:
        need("**" in first and "**" in second and len(first) > 80 and len(second) > 80, f"Worked swaps too thin Day {day}")
        need(bool(move.strip()), f"Teacher move missing Day {day}")


def assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    for stem, labels in (
        ("story-kit-guide", ("TENT", "POND", "PATH", "SCENE pocket", "TO SHOW A NEW VISITOR")),
        ("audience-feedback", ("AUDIENCE:", "FIRST WORDS", "ACTUAL LISTENER", "REVISED WORDS", "STILL UNCLEAR")),
        ("story-sample", ("SCENE 1", "SCENE 2", "SCENE 3", "SCENE 4", "Mira")),
        ("text-samples", ("STORY CARD STATION", "A map is on a mat.", "I can see a red cap.", "A cup is on a mat.")),
        ("portfolio-handover", ("ACTUAL SAMPLE 1", "ACTUAL SAMPLE 2", "NOT AVAILABLE", "MY NEXT TRY")),
    ):
        svg, pdf = ROOT / f"{stem}.svg", ROOT / f"{stem}.pdf"
        tree = ET.parse(svg).getroot()
        need(tree.attrib.get("width") == "210mm" and tree.attrib.get("height") == "297mm", f"{stem} SVG is not A4")
        need(tree.find("s:title", ns) is not None and tree.find("s:desc", ns) is not None, f"{stem} SVG lacks title/desc")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        need("Pages:           1" in info and "(A4)" in info, f"{stem} PDF is not single A4")
        extracted = " ".join(subprocess.check_output(["pdftotext", str(pdf), "-"], text=True).split())
        need(all(label in extracted for label in labels) and "CC BY 4.0" in extracted, f"{stem} PDF text/credit missing")
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8").lower()
    need(all(term in materials for term in ("tactile", "triangle roof", "large oval", "two separate parallel curved lines", "model / no actual feedback", "not available")), "Exact nonvisual and provenance alternatives absent")
    kit = ET.parse(ROOT / "story-kit-guide.svg").getroot()
    need(len(kit.findall("s:ellipse", ns)) == 1 and len([p for p in kit.findall("s:path", ns) if p.attrib.get("d", "").endswith("Z")]) == 1, "Kit outline icons differ from source")


def links() -> int:
    count = 0
    for path in ROOT.glob("*.md"):
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("https://", "http://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            need(target.exists(), f"Broken local link: {path.name} -> {url}")
            count += 1
    return count


def manifest_data() -> dict:
    files = sorted(
        path for path in ROOT.rglob("*")
        if path.is_file() and path.name != "manifest.json" and "__pycache__" not in path.parts and not path.name.startswith("_preview")
    )
    return {
        "pack_id": "subjectnest-acara-v9-foundation-english-t4-w39-40",
        "pack_version": "0.1.0-draft",
        "created_at": "2026-09-29",
        "review_status": "pending_human_educator_accessibility_local_syllabus_safety_privacy_and_classroom_review",
        "year_label": "Australian Curriculum Foundation Year / Queensland Prep",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(CODES),
        "curriculum_source": {
            "authority": "Australian Curriculum, Assessment and Reporting Authority",
            "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx",
            "retrieved_at": "2026-09-29",
            "source_hash": SOURCE_SHA,
            "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use",
        },
        "rights": "Original SubjectNest texts, scripts, checks and vector aids CC BY 4.0; ACARA retains its own terms.",
        "original_assets": [
            {"item_code_or_locator": str(path.relative_to(ROOT)), "sha256": hashlib.sha256(path.read_bytes()).hexdigest()}
            for path in files
        ],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    rows = official()
    lessons(rows)
    cards_and_checks()
    practice()
    assets()
    link_count = links()
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
    need(json.loads(manifest.read_text(encoding="utf-8")) == expected, "Hash manifest stale or missing")
    print(f"PASS Foundation English W39–40: 10x25m, 30 routes, 2 fresh checks/keys, 16 exact codes, 5 A4 aids, {link_count} local links, hashes")


if __name__ == "__main__":
    main()
