#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Verify authored Foundation English Weeks 37–38 and pinned ACARA source rows."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
ENGLISH = ROOT.parents[1]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = {"AC9EFLA05", "AC9EFLA06", "AC9EFLA07", "AC9EFLA08", "AC9EFLA09", "AC9EFLY01", "AC9EFLY02", "AC9EFLY05", "AC9EFLY06", "AC9EFLY07", "AC9EFLY10", "AC9EFLY12", "AC9EFLY13"}
PRINT = {"a", "an", "at", "can", "cap", "cup", "i", "in", "is", "it", "map", "mat", "on", "red", "see", "sit", "tap", "ten", "tin"}
HELD = {"sub", "gas", "hip", "tug"}
TEACHING = ("README.md", "LESSONS.md", "MATERIALS.md", "LEARNER-CARDS.md", "OPTIONAL-PRACTICE.md")


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def official() -> dict[str, dict]:
    workbook = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    data = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8"))
    need(hashlib.sha256(workbook.read_bytes()).hexdigest() == SOURCE_SHA == data["source_sha256"], "Official workbook/import hash differs")
    rows = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description" and r.get("code") in CODES}
    table: dict[str, tuple[int, str]] = {}
    for line in (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8").splitlines():
        match = re.match(r"^\| (AC9[A-Z0-9]+) \| (\d+) \| English · Foundation Year \| (.*?) \|$", line)
        if match:
            table[match.group(1)] = (int(match.group(2)), match.group(3))
    need(set(table) == CODES == set(rows), f"Crosswalk/source code mismatch: {set(table) ^ CODES}")
    for code, (source_row, description) in table.items():
        row = rows[code]
        need(row["attributes"]["level"] == "Foundation Year" and row["attributes"]["learning_area"] == "English", f"Wrong level/area: {code}")
        need(source_row == row["source_row"] and " ".join(description.split()) == " ".join(row["plain_text"].split()), f"Wrong source row/text: {code}")
    return rows


def lessons(rows: dict[str, dict]) -> None:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", content, re.MULTILINE))
    need([int(m.group(2)) for m in marks] == list(range(181, 191)), "Ten English days 181–190 required")
    for index, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (day - 1) // 5 + 1, f"Wrong week for Day {day}")
        body = content[mark.end(): marks[index + 1].start() if index + 1 < len(marks) else len(content)]
        times = [int(value) for value in re.findall(r"^\d+\. \*\*[^\n]*?\b(\d+) min\.", body, re.MULTILINE)]
        need(len(times) == 6 and sum(times) == 25, f"Day {day} needs six 25-minute steps, got {times}")
        need("**Goal:**" in body and "**Prepare:**" in body, f"Day {day} goal/preparation format missing")
        for code in set(re.findall(r"\bAC9[A-Z0-9]+\b", body)):
            need(code in rows, f"Unknown/non-Foundation lesson code: {code}")
    need("first independent" in content.lower() and "Adult reads" in content, "Assessment and adult-read boundary absent")


def cards_and_checks() -> None:
    cards = (ROOT / "LEARNER-CARDS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.MULTILINE))
    need([int(m.group(1)) for m in marks] == list(range(181, 191)), "Ten learner-card days required")
    for index, mark in enumerate(marks):
        body = cards[mark.end(): marks[index + 1].start() if index + 1 < len(marks) else len(cards)]
        lines = re.findall(r"^- \*\*[^*]+:\*\* .+?\*\*Child print:\*\* “([^”]+)”", body, re.MULTILINE)
        need(len(lines) == 3, f"Day {mark.group(1)} needs three same-target choice routes")
        for line in lines:
            words = set(re.findall(r"[a-z]+", line.lower()))
            need(words <= PRINT, f"Untaught child-print word Day {mark.group(1)}: {words - PRINT}")
    need("ASSESSMENT.md" not in cards and "TEACHER-KEY.md" not in cards, "Teacher-held content linked from learner page")
    teaching = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in TEACHING)
    prior = "\n".join(
        path.read_text(encoding="utf-8")
        for path in ENGLISH.rglob("*.md")
        if ROOT not in path.parents
        and path.name in TEACHING
        and ("term-4" not in path.parts or path.parent.name in {"weeks-31-32", "weeks-33-34", "weeks-35-36"})
    )
    assessment = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    for word in HELD:
        need(re.search(rf"\b{word}\b", teaching, re.IGNORECASE) is None, f"Held-out print item leaked into teaching: {word}")
        need(re.search(rf"\b{word}\b", prior, re.IGNORECASE) is None, f"Held-out print item appeared earlier: {word}")
        need(re.search(rf"\b{word}\b", assessment, re.IGNORECASE) is not None, f"Held-out item missing: {word}")
    for phrase in ("MASK: five", "FLAG: three", "DRUM: one", "two seconds", "MOCK DIAGRAM 1"):
        need(phrase.lower() not in teaching.lower(), f"Fresh check source leaked into teaching: {phrase}")
    need("first independent" in assessment.lower() and "not administered" in assessment.lower(), "Fresh-check first response/taught-point boundary absent")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    need(all(f"Day {day}" in assessment and f"Day {day}" in key for day in (185, 190)), "Two distinct held-out checks/keys required")
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8")
    need(all(term in materials for term in ("six broad circle cards", "four broad square cards", "two broad triangle cards", "two separate oval leaf outlines", "three separate oval leaf outlines", "Our school plant grew a new leaf overnight")), "Teaching source facts missing")
    need(all(term in assessment for term in ("MASK: five", "FLAG: three", "DRUM: one", "MOCK DIAGRAMS", "two seconds")), "Fresh source facts missing")


def practice() -> None:
    content = (ROOT / "OPTIONAL-PRACTICE.md").read_text(encoding="utf-8")
    rows = re.findall(r"^\| (18\d|190) \| (.+?) \| (.+?) \| (.+?) \|$", content, re.MULTILINE)
    need([int(row[0]) for row in rows] == list(range(181, 191)), "Optional practice needs two worked swaps on each of ten days")
    for day, first, second, move in rows:
        need("**" in first and "**" in second and len(first) > 80 and len(second) > 80, f"Worked swaps too thin Day {day}")
        need(bool(move.strip()), f"Teacher move missing Day {day}")


def assets() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    for stem, labels in (
        ("shape-tray-data", ("CIRCLE", "SQUARE", "TRIANGLE", "ONE OUTLINED SHAPE")),
        ("question-cards", ("QUESTION 1", "QUESTION 2", "QUESTION 3", "QUESTION 4")),
        ("data-report", ("QUESTION", "SOURCE", "WHAT THE DATA SAYS", "ANSWER WITH A CLUE", "WHAT THIS DOES NOT TELL US")),
        ("mock-plant-pair", ("MOCK DRAWING A", "MOCK DRAWING B", "Two oval leaf outlines", "Three oval leaf outlines")),
        ("fact-check-edit", ("FIRST CLAIM", "SOURCE", "WHAT IS ACTUALLY SHOWN", "REVISED CLAIM", "STILL TO CHECK")),
    ):
        svg, pdf = ROOT / f"{stem}.svg", ROOT / f"{stem}.pdf"
        tree = ET.parse(svg).getroot()
        need(tree.attrib.get("width") == "210mm" and tree.attrib.get("height") == "297mm", f"{stem} SVG is not A4")
        need(tree.find("s:title", ns) is not None and tree.find("s:desc", ns) is not None, f"{stem} SVG lacks title/desc")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        need("Pages:           1" in info and "(A4)" in info, f"{stem} PDF is not single A4")
        extracted = " ".join(subprocess.check_output(["pdftotext", str(pdf), "-"], text=True).split())
        need(all(label in extracted for label in labels) and "CC BY 4.0" in extracted, f"{stem} PDF text/credit missing")
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8").lower()
    need(all(term in materials for term in ("tactile", "circle: six", "two/three", "mock", "first claim")), "Exact nonvisual alternatives absent")
    shapes = ET.parse(ROOT / "shape-tray-data.svg").getroot()
    need(len(shapes.findall("s:circle", ns)) == 6, "Shape inventory circle count differs from source")
    need(len([r for r in shapes.findall("s:rect", ns) if r.attrib.get("width") == "14" and r.attrib.get("height") == "14"]) == 4, "Shape inventory square count differs from source")
    need(len([p for p in shapes.findall("s:path", ns) if p.attrib.get("d", "").endswith("Z")]) == 2, "Shape inventory triangle count differs from source")
    plant = ET.parse(ROOT / "mock-plant-pair.svg").getroot()
    leaves = plant.findall("s:ellipse", ns)
    need(len([e for e in leaves if float(e.attrib["cx"]) < 105]) == 2 and len([e for e in leaves if float(e.attrib["cx"]) > 105]) == 3, "Mock drawing leaf-outline counts differ from source")


def links() -> int:
    count = 0
    for path in ROOT.glob("*.md"):
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("https://", "http://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            need(target.exists(), f"Broken local link: {path.name} -> {url}")
            count += 1
    return count


def manifest_data() -> dict:
    files = sorted(
        path for path in ROOT.rglob("*")
        if path.is_file() and path.name != "manifest.json" and "__pycache__" not in path.parts and not path.name.startswith("_preview")
    )
    return {
        "pack_id": "subjectnest-acara-v9-foundation-english-t4-w37-38",
        "pack_version": "0.1.0-draft",
        "created_at": "2026-09-29",
        "review_status": "pending_human_educator_accessibility_local_syllabus_safety_privacy_and_classroom_review",
        "year_label": "Australian Curriculum Foundation Year / Queensland Prep",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(CODES),
        "curriculum_source": {
            "authority": "Australian Curriculum, Assessment and Reporting Authority",
            "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx",
            "retrieved_at": "2026-09-29",
            "source_hash": SOURCE_SHA,
            "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use",
        },
        "rights": "Original SubjectNest texts, scripts, checks and vector aids CC BY 4.0; ACARA retains its own terms.",
        "original_assets": [
            {"item_code_or_locator": str(path.relative_to(ROOT)), "sha256": hashlib.sha256(path.read_bytes()).hexdigest()}
            for path in files
        ],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    rows = official()
    lessons(rows)
    cards_and_checks()
    practice()
    assets()
    link_count = links()
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
    need(json.loads(manifest.read_text(encoding="utf-8")) == expected, "Hash manifest stale or missing")
    print(f"PASS Foundation English W37–38: 10x25m, 30 routes, 2 fresh checks/keys, 13 exact codes, 5 A4 aids, {link_count} local links, hashes")


if __name__ == "__main__":
    main()
