#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Verify authored Foundation English Weeks 33–34 and pinned ACARA source rows."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
ENGLISH = ROOT.parents[1]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = {"AC9EFLA05", "AC9EFLA06", "AC9EFLA07", "AC9EFLA08", "AC9EFLA09", "AC9EFLY01", "AC9EFLY02", "AC9EFLY05", "AC9EFLY06", "AC9EFLY07", "AC9EFLY10", "AC9EFLY12", "AC9EFLY13"}
PRINT = {"a", "an", "at", "can", "cap", "cup", "i", "in", "is", "it", "map", "mat", "on", "red", "see", "sit", "tap", "ten", "tin"}
HELD = {"cab", "tan", "hog", "fib"}
TEACHING = ("README.md", "LESSONS.md", "MATERIALS.md", "LEARNER-CARDS.md", "OPTIONAL-PRACTICE.md")


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def official() -> dict[str, dict]:
    workbook = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    data = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8"))
    need(hashlib.sha256(workbook.read_bytes()).hexdigest() == SOURCE_SHA == data["source_sha256"], "Official workbook/import hash differs")
    rows = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description" and r.get("code") in CODES}
    table: dict[str, tuple[int, str]] = {}
    for line in (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8").splitlines():
        match = re.match(r"^\| (AC9[A-Z0-9]+) \| (\d+) \| English · Foundation Year \| (.*?) \|$", line)
        if match:
            table[match.group(1)] = (int(match.group(2)), match.group(3))
    need(set(table) == CODES == set(rows), f"Crosswalk/source code mismatch: {set(table) ^ CODES}")
    for code, (source_row, description) in table.items():
        row = rows[code]
        need(row["attributes"]["level"] == "Foundation Year" and row["attributes"]["learning_area"] == "English", f"Wrong level/area: {code}")
        need(source_row == row["source_row"] and " ".join(description.split()) == " ".join(row["plain_text"].split()), f"Wrong source row/text: {code}")
    return rows


def lessons(rows: dict[str, dict]) -> None:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", content, re.MULTILINE))
    need([int(m.group(2)) for m in marks] == list(range(161, 171)), "Ten English days 161–170 required")
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (day - 1) // 5 + 1, f"Wrong week for Day {day}")
        body = content[mark.end(): marks[index + 1].start() if index + 1 < len(marks) else len(content)]
        times = [int(value) for value in re.findall(r"^\d+\. \*\*[^\n]*?\b(\d+) min\.", body, re.MULTILINE)]
        need(len(times) == 6 and sum(times) == 25, f"Day {day} needs six 25-minute steps, got {times}")
        need("**Goal:**" in body and "**Prepare:**" in body, f"Day {day} goal/preparation format missing")
        for code in set(re.findall(r"\bAC9[A-Z0-9]+\b", body)):
            need(code in rows, f"Unknown/non-Foundation lesson code: {code}")
    need("first independent" in content.lower() and "Adult reads" in content, "Assessment and adult-read boundary absent")


def cards_and_checks() -> None:
    cards = (ROOT / "LEARNER-CARDS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need([int(m.group(1)) for m in marks] == list(range(161, 171)), "Ten learner-card days required")
    for index, mark in enumerate(marks):
        body = cards[mark.end(): marks[index + 1].start() if index + 1 < len(marks) else len(cards)]
        lines = re.findall(r"^- \*\*[^*]+:\*\* .+?\*\*Child print:\*\* “([^”]+)”", body, re.MULTILINE)
        need(len(lines) == 3, f"Day {mark.group(1)} needs three same-target choice routes")
        for line in lines:
            words = set(re.findall(r"[a-z]+", line.lower()))
            need(words <= PRINT, f"Untaught child-print word Day {mark.group(1)}: {words - PRINT}")
    need("ASSESSMENT.md" not in cards and "TEACHER-KEY.md" not in cards, "Teacher-held content linked from learner page")
    teaching = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in TEACHING)
    prior = "\n".join(
        path.read_text(encoding="utf-8")
        for path in ENGLISH.rglob("*.md")
        if ROOT not in path.parents
        and path.name in TEACHING
        and ("term-4" not in path.parts or "weeks-31-32" in path.parts)
    )
    assessment = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    for word in HELD:
        need(re.search(rf"\b{word}\b", teaching, re.IGNORECASE) is None, f"Held-out print item leaked into teaching: {word}")
        need(re.search(rf"\b{word}\b", prior, re.IGNORECASE) is None, f"Held-out print item appeared earlier: {word}")
        need(re.search(rf"\b{word}\b", assessment, re.IGNORECASE) is not None, f"Held-out item missing: {word}")
    need(re.search(r"\bMalo\b", teaching, re.IGNORECASE) is None, "Fresh check character leaked into teaching")
    need("first independent" in assessment and "not administered" in assessment, "Fresh-check first response/taught-point boundary absent")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    need(all(f"Day {day}" in assessment and f"Day {day}" in key for day in (165, 170)), "Two distinct held-out checks/keys required")
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8")
    need(all(term in materials for term in ("three separate oval leaf outlines", "two separate narrow leaf outlines", "tissue tore", "card stayed whole", "same disk")), "Teaching source facts missing")
    need(all(term in assessment for term in ("four round leaf outlines", "three pointed leaf outlines", "five slow counts", "heavier star")), "Fresh source facts missing")


def assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    for stem, labels in (
        ("plant-observation", ("DRAWING P", "DRAWING Q", "3 oval leaves", "2 narrow leaves")),
        ("information-page", ("TITLE", "SOURCE", "DETAIL 1", "STILL TO FIND OUT")),
        ("roof-test", ("TRY 1", "TRY 2", "tissue tore", "card stayed whole")),
        ("explanation-cards", ("EXPLANATION 1", "EXPLANATION 2", "always best", "Another load")),
        ("claim-evidence-question", ("CLAIM", "SOURCE OR ACTUAL OBSERVATION", "STILL TO TEST")),
    ):
        svg, pdf = ROOT / f"{stem}.svg", ROOT / f"{stem}.pdf"
        tree = ET.parse(svg).getroot()
        need(tree.attrib.get("width") == "210mm" and tree.attrib.get("height") == "297mm", f"{stem} SVG is not A4")
        need(tree.find("s:title", ns) is not None and tree.find("s:desc", ns) is not None, f"{stem} SVG lacks title/desc")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        need("Pages:           1" in info and "(A4)" in info, f"{stem} PDF is not single A4")
        extracted = " ".join(subprocess.check_output(["pdftotext", str(pdf), "-"], text=True).split())
        need(all(label in extracted for label in labels) and "CC BY 4.0" in extracted, f"{stem} PDF text/credit missing")
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8").lower()
    need(all(term in materials for term in ("linear", "tactile", "source", "still to test", "colour")), "Exact nonvisual alternatives absent")


def links() -> int:
    count = 0
    for path in ROOT.glob("*.md"):
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("https://", "http://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            need(target.exists(), f"Broken local link: {path.name} -> {url}")
            count += 1
    return count


def manifest_data() -> dict:
    files = sorted(
        path for path in ROOT.rglob("*")
        if path.is_file() and path.name != "manifest.json" and "__pycache__" not in path.parts and not path.name.startswith("_preview")
    )
    return {
        "pack_id": "subjectnest-acara-v9-foundation-english-t4-w33-34",
        "pack_version": "0.1.0-draft",
        "created_at": "2026-09-29",
        "review_status": "pending_human_educator_accessibility_local_syllabus_safety_privacy_and_classroom_review",
        "year_label": "Australian Curriculum Foundation Year / Queensland Prep",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(CODES),
        "curriculum_source": {
            "authority": "Australian Curriculum, Assessment and Reporting Authority",
            "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx",
            "retrieved_at": "2026-09-29",
            "source_hash": SOURCE_SHA,
            "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use",
        },
        "rights": "Original SubjectNest texts, scripts, checks and vector aids CC BY 4.0; ACARA retains its own terms.",
        "original_assets": [
            {"item_code_or_locator": str(path.relative_to(ROOT)), "sha256": hashlib.sha256(path.read_bytes()).hexdigest()}
            for path in files
        ],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    rows = official()
    lessons(rows)
    cards_and_checks()
    assets()
    link_count = links()
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
    need(json.loads(manifest.read_text(encoding="utf-8")) == expected, "Hash manifest stale or missing")
    print(f"PASS Foundation English W33–34: 10x25m, 30 routes, 2 fresh checks/keys, 13 exact codes, 5 A4 aids, {link_count} local links, hashes")


if __name__ == "__main__":
    main()
