#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Verify Foundation English W27–28 and integrated source/rights boundaries."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
FOUNDATION = ROOT.parents[2]
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CROSSWALK = ROOT / "CURRICULUM-CROSSWALK.md"
EXPECTED = set("AC9EFLA07 AC9EFLY01 AC9EFLY02 AC9EFLY04 AC9EFLY05 AC9EFLY06 AC9EFLY07 AC9EFLY10 AC9EFLY12 AC9EFLY13 AC9MFM01 AC9MFST01 AC9SFU03 AC9SFI01 AC9SFI02 AC9SFI03 AC9SFI04 AC9HSFS01 AC9HSFS03 AC9HSFS04".split())
PRINT = set("a bag can cap cup i in is map mat on red see ten tin".split())
HELD = set("peg rag bun hem cub rib pod gum".split())


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def official() -> dict[str, dict]:
    workbook = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    data = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8"))
    need(hashlib.sha256(workbook.read_bytes()).hexdigest() == SOURCE_SHA == data["source_sha256"], "Pinned workbook/import hash mismatch")
    need(data["framework"] == "Australian Curriculum Version 9.0", "Unexpected curriculum version")
    rows = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description"}
    table = {}
    for line in CROSSWALK.read_text(encoding="utf-8").splitlines():
        m = re.match(r"^\| (AC9[A-Z0-9]+) \| (\d+) \| (.+?) · Foundation Year \| (.*?) \|", line)
        if m:
            table[m.group(1)] = (int(m.group(2)), m.group(3), m.group(4))
    need(set(table) == EXPECTED, f"Crosswalk codes differ: {set(table) ^ EXPECTED}")
    for code, (row_number, area, description) in table.items():
        row = rows[code]
        need(row["source_row"] == row_number and row["attributes"]["learning_area"] == area, f"Wrong area/row: {code}")
        need(row["attributes"]["level"] == "Foundation Year" and " ".join(row["plain_text"].split()) == description, f"Wrong level/wording: {code}")
    return rows


def check_lessons(rows: dict[str, dict]) -> None:
    text = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", text, re.M))
    days = [int(m.group(2)) for m in marks]
    need(days == list(range(131, 141)), f"Ten English days expected: {days}")
    for i, m in enumerate(marks):
        day = int(m.group(2))
        need(int(m.group(1)) == (day - 1) // 5 + 1, f"Wrong week for Day {day}")
        body = text[m.end():marks[i+1].start() if i+1 < len(marks) else len(text)]
        steps = [int(x) for x in re.findall(r"^\d+\. \*\*[^\n]*?\b(\d+) min\.", body, re.M)]
        need(len(steps) == 6 and sum(steps) == 25, f"Day {day}: expected six timed 25-minute steps, got {steps}")
        for code in set(re.findall(r"\bAC9[A-Z0-9]+\b", body)):
            need(code in rows and rows[code]["attributes"]["level"] == "Foundation Year", f"Unknown/non-Foundation code {code}")
    need("adult reads" in text.lower() and "first independent" in text.lower(), "Print/evidence boundary absent")


def check_cards_and_checks() -> None:
    cards = (ROOT / "LEARNER-CARDS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", cards, re.M))
    need([int(m.group(1)) for m in marks] == list(range(131, 141)), "Learner card days missing")
    for i, m in enumerate(marks):
        body = cards[m.end():marks[i+1].start() if i+1 < len(marks) else len(cards)]
        lines = re.findall(r"^\- \*\*[^*]+:\*\* .*?\*\*Child print:\*\* `([^`]+)`", body, re.M)
        need(len(lines) == 3, f"Day {m.group(1)} needs three child choices")
        for line in lines:
            words = re.findall(r"[A-Za-z]+", line.lower())
            need(set(words) <= PRINT, f"Untaught inventory in Day {m.group(1)} child line: {set(words)-PRINT}")
    need(len(re.findall(r"\*\*Child print:\*\*", cards)) == 30, "Expected 30 child lines")
    teaching = "\n".join((ROOT / name).read_text(encoding="utf-8") for name in ("README.md", "LESSONS.md", "MATERIALS.md", "LEARNER-CARDS.md"))
    prior = "\n".join(p.read_text(encoding="utf-8") for p in (FOUNDATION / "english").rglob("*.md") if ROOT not in p.parents)
    for word in HELD:
        need(not re.search(rf"\b{word}\b", teaching, re.I), f"Held-out word leaked into teaching: {word}")
        need(not re.search(rf"\b{word}\b", prior, re.I), f"Held-out word appeared in prior English: {word}")
    checks = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    need(all(re.search(rf"\b{word}\b", checks) for word in HELD), "Held-out teacher items missing")
    need("ASSESSMENT.md" not in cards and "key" not in cards.lower(), "Learner card references teacher key")
    need("pass" in checks.lower() and "first independent" in checks.lower(), "Assessment privacy/first response absent")


def check_integrated(rows: dict[str, dict]) -> None:
    for week, first in ((27, 131), (28, 136)):
        paths = sorted((FOUNDATION / "term-3").glob(f"week-{week:02d}-*.md"))
        need(len(paths) == 1, f"Expected one integrated Week {week}")
        content = paths[0].read_text(encoding="utf-8")
        table_rows = [line for line in content.splitlines() if re.match(r"^\| \*\*\d{3} ·", line)]
        need(len(table_rows) == 5, f"Week {week}: five integrated blocks expected")
        for day, line in zip(range(first, first+5), table_rows):
            need(re.match(rf"^\| \*\*{day} ·", line) is not None, f"Integrated Day {day} missing")
            times = [int(x) for x in re.findall(r"\*\*(\d+) min\*\*", line)]
            need(len(times) == 5 and sum(times) == 35, f"Integrated Day {day} timings {times}")
        for code in set(re.findall(r"\bAC9[A-Z0-9]+\b", content)):
            need(code in rows and rows[code]["attributes"]["level"] == "Foundation Year", f"Integrated invalid code: {code}")


def check_links() -> int:
    paths = list(ROOT.rglob("*.md")) + sorted((FOUNDATION / "term-3").glob("week-27-*.md")) + sorted((FOUNDATION / "term-3").glob("week-28-*.md"))
    count = 0
    for path in paths:
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("http://", "https://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            need(target.exists(), f"Broken link: {path} -> {url}")
            count += 1
    return count


def check_aid() -> None:
    svg = ROOT / "fact-question-report.svg"
    pdf = ROOT / "fact-question-report.pdf"
    root = ET.parse(svg).getroot()
    ns = {"s": "http://www.w3.org/2000/svg"}
    need(root.attrib.get("width") == "210mm" and root.attrib.get("height") == "297mm", "Aid must be A4")
    need(root.find("s:title", ns) is not None and root.find("s:desc", ns) is not None, "Aid title/desc missing")
    need("(A4)" in subprocess.check_output(["pdfinfo", str(pdf)], text=True), "PDF not A4")
    pdf_text = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
    need(all(x in pdf_text for x in ("FACT", "QUESTION", "REPORT", "SOURCE NAMED", "CC BY 4.0")), "PDF text missing")
    materials = (ROOT / "MATERIALS.md").read_text(encoding="utf-8").lower()
    need(all(x in materials for x in ("tactile", "fact", "question", "ask", "listen", "report")), "Text alternative missing")


def make_manifest() -> dict:
    files = sorted(p for p in ROOT.rglob("*") if p.is_file() and p.name != "manifest.json" and "__pycache__" not in p.parts)
    return {
        "pack_id": "subjectnest-acara-v9-foundation-english-t3-w27-28",
        "pack_version": "0.1.0-draft", "created_at": "2026-09-29",
        "review_status": "pending_human_educator_accessibility_local_syllabus_privacy_and_classroom_review",
        "year_label": "Australian Curriculum Foundation Year / Queensland Prep",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(c for c in EXPECTED if c.startswith("AC9EF")),
        "curriculum_source": {"authority": "Australian Curriculum, Assessment and Reporting Authority", "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx", "retrieved_at": "2026-09-29", "source_hash": SOURCE_SHA, "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use"},
        "rights": "Original SubjectNest text, interview scripts, checks and vector aid CC BY 4.0; ACARA/AERO retain own terms.",
        "original_assets": [{"item_code_or_locator": str(p.relative_to(ROOT)), "sha256": hashlib.sha256(p.read_bytes()).hexdigest()} for p in files],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    rows = official()
    check_lessons(rows)
    check_cards_and_checks()
    check_integrated(rows)
    check_aid()
    manifest = ROOT / "manifest.json"
    expected = make_manifest()
    if args.write_manifest:
        manifest.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
    need(json.loads(manifest.read_text(encoding="utf-8")) == expected, "Manifest stale/missing")
    links = check_links()
    print(f"PASS Foundation English W27–28: 10x25m days, 30 child choices, 2 held-out checks, 10x35m integrated blocks, {len(EXPECTED)} exact Foundation codes, {links} local links, A4/text aid, hashes")


if __name__ == "__main__":
    main()
