# SPDX-License-Identifier: Apache-2.0
"""Verify Year 5 English W03–04 exact curriculum, lesson, checks and assets."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
CODES = {"AC9E5LA01", "AC9E5LA02", "AC9E5LA03", "AC9E5LA04", "AC9E5LA05", "AC9E5LA06", "AC9E5LA07", "AC9E5LA08", "AC9E5LE02", "AC9E5LE03", "AC9E5LE04", "AC9E5LE05", "AC9E5LY02", "AC9E5LY03", "AC9E5LY04", "AC9E5LY05", "AC9E5LY06", "AC9E5LY07", "AC9E5LY09"}
TEACHING = ("README.md", "TEXTS.md", "LESSONS.md", "LEARNER-COPY.md", "OPTIONAL-PRACTICE.md", "HOME-EXTENSIONS.md", "PRINT-ALTERNATIVES.md")


def need(condition: bool, message: str) -> None:
    if not condition:
        raise AssertionError(message)


def official() -> dict[str, dict]:
    workbook = STUDIO / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    data = json.loads((STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8"))
    need(hashlib.sha256(workbook.read_bytes()).hexdigest() == SOURCE_SHA == data["source_sha256"], "Pinned workbook/import hash mismatch")
    need(data["framework"] == "Australian Curriculum Version 9.0", "Framework version mismatch")
    rows = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description"}
    table = {}
    for line in (ROOT / "CURRICULUM-CROSSWALK.md").read_text(encoding="utf-8").splitlines():
        m = re.match(r"^\| (AC9E5[A-Z0-9]+) \| (\d+) \| Year 5 \| (.*?) \|$", line)
        if m:
            table[m.group(1)] = (int(m.group(2)), m.group(3))
    need(set(table) == CODES, f"Exact crosswalk codes differ: {set(table) ^ CODES}")
    for code, (source_row, description) in table.items():
        r = rows[code]
        need(r["attributes"]["level"] == "Year 5" and r["attributes"]["learning_area"] == "English", f"Wrong level/area: {code}")
        need(r["source_row"] == source_row and " ".join(r["plain_text"].split()) == description, f"Wrong source row/wording: {code}")
    return rows


def lesson_structure(rows: dict[str, dict]) -> None:
    content = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Week (\d+) · Day (\d+)\b", content, re.MULTILINE))
    need([int(m.group(2)) for m in marks] == list(range(11, 21)), "Ten Day 11–20 lesson headers required")
    for i, mark in enumerate(marks):
        day = int(mark.group(2))
        need(int(mark.group(1)) == (3 if day <= 15 else 4), f"Wrong week for Day {day}")
        body = content[mark.end():marks[i+1].start() if i+1 < len(marks) else len(content)]
        steps = [int(x) for x in re.findall(r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*", body, re.MULTILINE)]
        need(len(steps) == 6 and sum(steps) == 25, f"Day {day}: six timed 25-minute steps required, got {steps}")
        need("**Goal:**" in body and "**Prepare:**" in body and ("**Exit" in body or "**Close" in body), f"Day {day} lacks goal/prep/exit")
        for code in set(re.findall(r"\bAC9E5[A-Z0-9]+\b", body)):
            need(code in rows and code in CODES, f"Invalid lesson code {code}")
    need("first independent" in content and "not scored as independent print reading" in content, "Access/first-response boundary absent")


def learner_and_checks() -> None:
    child = (ROOT / "LEARNER-COPY.md").read_text(encoding="utf-8")
    marks = list(re.finditer(r"^## Day (\d+)\b", child, re.MULTILINE))
    need([int(m.group(1)) for m in marks] == list(range(11, 21)), "Ten child-card days required")
    for i, mark in enumerate(marks):
        body = child[mark.end():marks[i+1].start() if i+1 < len(marks) else len(child)]
        need(len(re.findall(r"^- \*\*", body, re.MULTILINE)) == 3, f"Day {mark.group(1)} needs three choices")
    need("after the new task" in child and "TEACHER-KEY" not in child and "ASSESSMENT.md" not in child, "Private check linked/leaked in child copy")
    practice = (ROOT / "OPTIONAL-PRACTICE.md").read_text(encoding="utf-8")
    need([int(x) for x in re.findall(r"^\| (\d{2}) \|", practice, re.MULTILINE)] == list(range(11, 21)), "Ten daily extra-practice rows required")
    home = (ROOT / "HOME-EXTENSIONS.md").read_text(encoding="utf-8")
    need([int(x) for x in re.findall(r"^\| (\d{2}) \|", home, re.MULTILINE)] == list(range(11, 21)), "Ten optional home routes required")
    teaching = "\n".join((ROOT / n).read_text(encoding="utf-8") for n in TEACHING)
    for phrase in ("Paper-plant label table", "Tali slid a curling map", "My certainty had been a locked door"):
        need(phrase not in teaching, f"Held-out source leaked into teaching: {phrase}")
    check = (ROOT / "ASSESSMENT.md").read_text(encoding="utf-8")
    key = (ROOT / "TEACHER-KEY.md").read_text(encoding="utf-8")
    need(all(s in check for s in ("Paper-plant label table", "Nine labels", "eleven", "Tali slid a curling map", "first response")), "Fresh sources/evidence capture incomplete")
    need(all(s in key for s in ("20 labels", "not 20", "unfair start", "locked door", "component")), "Teacher key/response moves incomplete")
    need("read-aloud" in check.lower() and "not" in check.lower() and "independent reading" in check.lower(), "Reading access boundary missing")


def aids() -> None:
    ns = {"s": "http://www.w3.org/2000/svg"}
    specs = {
        "source-scope-mat": ("WHO / ROLE", "LIMIT / STILL UNKNOWN"),
        "panel-sequence": ("PERMISSION", "AFTER CAPTIONS"),
        "viewpoint-evidence-mat": ("NARRATOR ASSUMED", "STILL UNKNOWN"),
    }
    for stem, phrases in specs.items():
        svg, pdf = ROOT / f"{stem}.svg", ROOT / f"{stem}.pdf"
        root = ET.parse(svg).getroot()
        need(root.attrib.get("width") == "210mm" and root.attrib.get("height") == "297mm" and root.attrib.get("viewBox") == "0 0 210 297", f"{stem} not A4 portrait")
        need(root.attrib.get("role") == "img" and root.attrib.get("aria-labelledby") == "title desc", f"{stem} accessible SVG metadata missing")
        need(root.find("s:title", ns) is not None and root.find("s:desc", ns) is not None, f"{stem} title/desc absent")
        need("(A4)" in subprocess.check_output(["pdfinfo", str(pdf)], text=True), f"{stem} PDF not A4")
        pdf_text = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        need(all(s in pdf_text for s in phrases) and "CC BY 4.0" in pdf_text, f"{stem} PDF text/credit missing")
        fonts = subprocess.check_output(["pdffonts", str(pdf)], text=True)
        need("DejaVuSans" in fonts, f"{stem} embedded font unexpected")
    notice = ROOT.parents[2] / "term-1/weeks-01-02/print/FONT-RIGHTS.md"
    need(notice.exists() and "Bitstream Vera" in notice.read_text(encoding="utf-8"), "DejaVu font rights notice missing")
    alt = (ROOT / "PRINT-ALTERNATIVES.md").read_text(encoding="utf-8")
    need(all(s in alt for s in ("WHO / ROLE", "Hala asked first", "NARRATOR ASSUMED", "Tactile", "not asserted to be tagged")), "Text/tactile aid alternatives incomplete")


def links() -> int:
    count = 0
    for path in ROOT.glob("*.md"):
        for url in re.findall(r"(?<!!)\[[^]]+\]\(([^)]+)\)", path.read_text(encoding="utf-8")):
            if url.startswith(("https://", "http://", "mailto:", "#")):
                continue
            target = (path.parent / unquote(url.partition("#")[0])).resolve()
            need(target.exists(), f"Broken local link {path.name}: {url}")
            count += 1
    return count


def rights(path: Path) -> str:
    if path.suffix == ".pdf":
        return "Original SubjectNest design CC BY 4.0; embedded DejaVu font subsets under separate Bitstream Vera licence notice"
    if path.suffix == ".py":
        return "Apache-2.0 verification code"
    return "CC BY 4.0; original SubjectNest material"


def manifest_data() -> dict:
    files = sorted(p for p in ROOT.iterdir() if p.is_file() and p.name != "manifest.json")
    return {
        "pack_id": "subjectnest-acara-v9-year-5-english-t1-w03-04",
        "pack_version": "0.1.0-draft", "created_at": "2026-09-29",
        "review_status": "pending_human_educator_accessibility_local_syllabus_privacy_and_classroom_review",
        "year_label": "Australian Curriculum Year 5 / Queensland Year 5",
        "alignment_status": "proposed_partial_not_authority_approved",
        "curriculum_codes": sorted(CODES),
        "curriculum_source": {"authority": "Australian Curriculum, Assessment and Reporting Authority", "source_url": "https://www.australiancurriculum.edu.au/content/dam/en/curriculum/ac-version-9/downloads/curriculum-workbook.xlsx", "retrieved_at": "2026-09-29", "source_hash": SOURCE_SHA, "terms_url": "https://www.australiancurriculum.edu.au/copyright-and-terms-of-use"},
        "rights": "Original SubjectNest texts/checks/vector designs CC BY 4.0; ACARA and embedded DejaVu fonts retain separate terms.",
        "original_assets": [{"item_code_or_locator": p.name, "sha256": hashlib.sha256(p.read_bytes()).hexdigest(), "rights": rights(p)} for p in files],
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    rows = official()
    lesson_structure(rows)
    learner_and_checks()
    aids()
    link_count = links()
    expected = manifest_data()
    manifest = ROOT / "manifest.json"
    if args.write_manifest:
        manifest.write_text(json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
    need(json.loads(manifest.read_text(encoding="utf-8")) == expected, "Manifest stale/missing")
    print(f"PASS Year 5 English W03–04: 10x25m days, 30 learner routes, 10 follow-ups, 2 held-out checks, {len(CODES)} exact codes, 3 A4 SVG/PDF, {link_count} links, hashes")


if __name__ == "__main__":
    main()
