#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Verify authored Foundation English W29–30, source rows and paired guides."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import wave
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
FOUNDATION = ROOT.parents[2]
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = set("AC9EFLA05 AC9EFLA06 AC9EFLA07 AC9EFLE04 AC9EFLY02 AC9EFLY05 AC9EFLY06 AC9EFLY07 AC9EFLY09 AC9EFLY10 AC9EFLY12 AC9EFLY13 AC9MFA01 AC9MFN01 AC9MFN03 AC9AMUFE01 AC9AMUFD01 AC9AMUFC01 AC9AMUFP01 AC9AMAFE01 AC9AMAFD01 AC9AMAFC01 AC9AMAFP01 AC9AVAFE01 AC9AVAFD01 AC9AVAFC01 AC9AVAFP01 AC9TDEFK01 AC9TDEFP01 AC9SFU03 AC9SFI01 AC9SFI02 AC9SFI03 AC9SFI04 AC9SFI05".split())
PRINT = set("a an at can cap cup i in is it map mat on red see sit tap ten tin".split())
HELD = set("dim ram bop pup".split())
GUIDES = [FOUNDATION / "term-3/week-29-rhythm-picture-message.md", FOUNDATION / "term-3/week-30-a-stronger-design.md"]


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def official() -> dict[str, dict]:
    workbook = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    data = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8"))
    need(hashlib.sha256(workbook.read_bytes()).hexdigest() == SOURCE_SHA == data["source_sha256"], "Pinned official workbook/import mismatch")
    rows = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description"}
    table = {}
    for line in (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8").splitlines():
        m = re.match(r"^\| (AC9[A-Z0-9]+) \| (\d+) \| (.+?) · Foundation Year \| (.*?) \|$", line)
        if m:
            table[m.group(1)] = (int(m.group(2)), m.group(3), m.group(4))
    need(set(table) == CODES, f"Crosswalk codes differ: {set(table) ^ CODES}")
    for code, (source_row, area, description) in table.items():
        row = rows[code]
        attrs = row["attributes"]
        expected_area = attrs["learning_area"] + (f" / {attrs['subject']}" if attrs.get("subject") and attrs["subject"] != attrs["learning_area"] else "")
        need(attrs["level"] == "Foundation Year" and area == expected_area, f"Wrong area/level {code}")
        need(row["source_row"] == source_row and " ".join(row["plain_text"].split()) == description, f"Wrong source row/text {code}")
    return rows


def lessons(rows: dict[str, dict]) -> None:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", content, re.M))
    need([int(m.group(2)) for m in marks] == list(range(141, 151)), "Ten English days 141–150 required")
    for i, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (day - 1) // 5 + 1, f"Wrong week for Day {day}")
        body = content[mark.end():marks[i+1].start() if i+1 < len(marks) else len(content)]
        times = [int(x) for x in re.findall(r"^\d+\. \*\*[^\n]*?\b(\d+) min\.", body, re.M)]
        need(len(times) == 6 and sum(times) == 25, f"Day {day} needs six timed 25-minute steps: {times}")
        for code in re.findall(r"\bAC9[A-Z0-9]+\b", body):
            need(code in rows and code in CODES, f"Unknown/non-Foundation lesson code {code}")
    need("first independent" in content and "Adult reads" in content, "Print/first-response boundary missing")


def cards_and_checks() -> None:
    cards = (ROOT / "LEARNER-CARDS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.M))
    need([int(m.group(1)) for m in marks] == list(range(141, 151)), "Ten child-card days required")
    for i, mark in enumerate(marks):
        body = cards[mark.end():marks[i+1].start() if i+1 < len(marks) else len(cards)]
        lines = re.findall(r"^- \*\*[^*]+:\*\* .*?\*\*Child print:\*\* `([^`]+)`", body, re.M)
        need(len(lines) == 3, f"Day {mark.group(1)} needs three child choices with print")
        for line in lines:
            words = set(re.findall(r"[a-z]+", line.lower()))
            need(words <= PRINT, f"Untaught child print in Day {mark.group(1)}: {words - PRINT}")
    need("ASSESSMENT.md" not in cards and "TEACHER-KEY" not in cards, "Teacher copy leaked into child cards")
    teaching = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in ("README.md", "LESSONS.md", "MATERIALS.md", "LEARNER-CARDS.md", "OPTIONAL-PRACTICE.md"))
    prior = "\n".join(p.read_text(encoding="utf-8") for p in (FOUNDATION / "english").rglob("*.md") if ROOT not in p.parents)
    check = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    for word in HELD:
        need(not re.search(rf"\b{word}\b", teaching, re.I), f"Held-out word leaked into teaching: {word}")
        need(not re.search(rf"\b{word}\b", prior, re.I), f"Held-out word already used in prior teaching: {word}")
        need(re.search(rf"\b{word}\b", check) is not None, f"Held-out item missing: {word}")
    need("first independent" in check and "not administered" in check, "Assessment print/first-response safeguard absent")


def integrated(rows: dict[str, dict]) -> None:
    for week, first, path in [(29, 141, GUIDES[0]), (30, 146, GUIDES[1])]:
        content = path.read_text(encoding="utf-8")
        table_rows = [line for line in content.splitlines() if re.match(r"^\| \*\*\d{3} ·", line)]
        need(len(table_rows) == 5, f"Week {week} needs five integrated blocks")
        for day, line in zip(range(first, first + 5), table_rows):
            need(re.match(rf"^\| \*\*{day} ·", line) is not None, f"Integrated Day {day} absent")
            times = [int(x) for x in re.findall(r"\*\*(\d+) min\*\*", line)]
            need(len(times) == 5 and sum(times) == 35, f"Integrated Day {day}: {times}")
        for code in set(re.findall(r"\bAC9[A-Z0-9]+\b", content)):
            need(code in rows and code in CODES, f"Invalid integrated code {code}")


def assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    for stem in ("beat-pause-message", "before-change-because"):
        svg, pdf = ROOT / f"{stem}.svg", ROOT / f"{stem}.pdf"
        root = ET.parse(svg).getroot()
        need(root.attrib.get("width") == "210mm" and root.attrib.get("height") == "297mm", f"{stem} not A4")
        need(root.find("s:title", ns) is not None and root.find("s:desc", ns) is not None, f"{stem} accessibility metadata missing")
        need("(A4)" in subprocess.check_output(["pdfinfo", str(pdf)], text=True), f"{stem} PDF not A4")
        need("CC BY 4.0" in subprocess.check_output(["pdftotext", str(pdf), "-"], text=True), f"{stem} PDF credit absent")
    with wave.open(str(ROOT / "two-taps-rest.wav"), "rb") as audio:
        need(audio.getnchannels() == 1 and audio.getframerate() == 16000 and audio.getnframes() == 72000, "Cue duration/channel differs")
        frames = audio.readframes(audio.getnframes())
        need(any(frames), "Cue is silent")
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8").lower()
    need(all(term in materials for term in ("transcript", "tactile", "beat 1", "because", "no colour")), "Accessible aid alternatives missing")


def links() -> int:
    count = 0
    for path in list(ROOT.rglob("*.md")) + GUIDES:
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("https://", "http://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            need(target.exists(), f"Broken local link: {path.name} -> {url}")
            count += 1
    return count


def manifest_data() -> dict:
    files = sorted(p for p in ROOT.rglob("*") if p.is_file() and p.name != "manifest.json" and "__pycache__" not in p.parts)
    return {
        "pack_id": "subjectnest-acara-v9-foundation-english-t3-w29-30",
        "pack_version": "0.1.0-draft", "created_at": "2026-09-29",
        "review_status": "pending_human_educator_accessibility_local_syllabus_safety_privacy_and_classroom_review",
        "year_label": "Australian Curriculum Foundation Year / Queensland Prep",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(c for c in CODES if c.startswith("AC9EF")),
        "curriculum_source": {"authority": "Australian Curriculum, Assessment and Reporting Authority", "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx", "retrieved_at": "2026-09-29", "source_hash": SOURCE_SHA, "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use"},
        "rights": "Original SubjectNest texts, audio, lesson scripts, checks and vector aids CC BY 4.0; ACARA retains own terms.",
        "original_assets": [{"item_code_or_locator": str(p.relative_to(ROOT)), "sha256": hashlib.sha256(p.read_bytes()).hexdigest()} for p in files],
        "related_integrated_guides": [{"item_code_or_locator": str(p.relative_to(FOUNDATION)), "sha256": hashlib.sha256(p.read_bytes()).hexdigest()} for p in GUIDES],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    rows = official()
    lessons(rows)
    cards_and_checks()
    integrated(rows)
    assets()
    link_count = links()
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
    need(json.loads(manifest.read_text(encoding="utf-8")) == expected, "Manifest stale/missing")
    print(f"PASS Foundation English W29–30: 10x25m, 30 choices, 2 held-out checks, 10x35m integrated, {len(CODES)} exact codes, 2 A4 aids + WAV, {link_count} local links, hashes")


if __name__ == "__main__":
    main()
