# SPDX-License-Identifier: Apache-2.0
"""Read-only source, lesson, check, accessible-print and hash audit for HASS W19–20."""

from __future__ import annotations

import argparse
import ast
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
STUDIO = ROOT.parents[4]
SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
ROWS = {
    "AC9HSFK01": 1213,
    "AC9HSFK02": 1217,
    "AC9HSFK03": 1221,
    "AC9HSFK04": 1226,
    "AC9HSFS01": 1233,
    "AC9HSFS02": 1237,
    "AC9HSFS03": 1241,
    "AC9HSFS04": 1245,
    "AC9HSFS05": 1251,
}
HELD = {f"AC9HSFK0{i}" for i in range(1, 5)}
SKILLS = set(ROWS) - HELD
QCAA = "https://www.qcaa.qld.edu.au/p-10/aciq/version-9/learning-areas/p-10-humanities-and-social-sciences/hass"
ALIGN = "https://www.qcaa.qld.edu.au/downloads/aciqv9/humanities-and-social-sciences/curriculum/ac9_hass_prep_as_cd_alignment.pdf"
STEMS = (
    "story-sign-sketch",
    "ivo-maker-note",
    "sol-display-caption",
    "tala-visitor-note",
    "source-sort-mat",
    "explain-to-someone-mat",
    "check-a-pocket-card",
    "check-b-shelf-tag",
)
PHRASES = {
    "story-sign-sketch": (
        "one blue circle above three wavy lines",
        "does not name a maker",
    ),
    "ivo-maker-note": (
        "I made the paper sign",
        "I do not know whether",
        "every visitor used it",
    ),
    "sol-display-caption": ("I copied Ivo's note", "Everyone found the table"),
    "tala-visitor-note": ("after asking Ivo", "I cannot speak for other visitors"),
    "source-sort-mat": ("OBJECT DRAWING", "MAKER'S NOTE", "VISITOR'S NOTE"),
    "explain-to-someone-mat": ("SOURCE SAYS", "PERSON SAYS", "STILL OPEN"),
    "check-a-pocket-card": (
        "orange square",
        "two dots",
        "I do not know whether anyone else picked it",
        "Every player picked",
    ),
    "check-b-shelf-tag": (
        "green diamond and three bars",
        "I asked Una where",
        "I do not know about anyone else",
    ),
}
EXPECTED = {
    91: {"AC9HSFS01", "AC9HSFS04"},
    92: {"AC9HSFS01", "AC9HSFS02", "AC9HSFS04"},
    93: {"AC9HSFS03", "AC9HSFS04"},
    94: set(SKILLS),
    95: set(SKILLS),
    96: {"AC9HSFS02", "AC9HSFS03", "AC9HSFS04"},
    97: {"AC9HSFS01", "AC9HSFS03", "AC9HSFS04"},
    98: {"AC9HSFS01", "AC9HSFS03", "AC9HSFS04", "AC9HSFS05"},
    99: set(SKILLS),
    100: set(SKILLS),
}
REQUIRED = {
    "README.md",
    "MATERIALS.md",
    "LESSONS.md",
    "LEARNER-CARDS.md",
    "PRACTICE-SWAPS.md",
    "FAMILY-CARER-OPTIONS.md",
    "STUDENT-CHECKS.md",
    "teacher/KEY-AND-NEXT.md",
    "CURRICULUM-CROSSWALK.md",
    "LOCAL-SOURCE-INSERT.md",
    "SOURCE-AND-RIGHTS.md",
    "RUN-THROUGH.md",
    "source-snapshot.json",
    "CODE-LICENSE.txt",
    "verify_pack.py",
    "print/generate_print.py",
    "print/TEXT-ALTERNATIVES.md",
    "print/DEJAVU-FONT-LICENSE.txt",
}
for stem in STEMS:
    REQUIRED.update((f"print/{stem}.svg", f"print/{stem}.pdf"))


def need(ok, message):
    if not ok:
        raise AssertionError(message)


def read(name):
    return (ROOT / name).read_text(encoding="utf-8")


def table_row(name, day, count):
    rows = re.findall(rf"^\| {day} \|(.+)$", read(name), re.MULTILINE)
    need(len(rows) == 1, f"{name}: Day {day} absent or duplicated")
    cells = [x.strip() for x in rows[0].strip().strip("|").split("|")]
    need(len(cells) == count and all(cells), f"{name}: Day {day} cell drift")
    return cells


def sources():
    workbook = (
        STUDIO
        / "research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx"
    )
    imported = json.loads(
        (STUDIO / "data/frameworks/acara-v9.json").read_text(encoding="utf-8")
    )
    snapshot = json.loads(read("source-snapshot.json"))
    cross = read("CURRICULUM-CROSSWALK.md")
    need(
        hashlib.sha256(workbook.read_bytes()).hexdigest()
        == imported["source_sha256"]
        == snapshot["workbook_sha256"]
        == SHA,
        "official workbook digest drift",
    )
    need(snapshot["content_rows"] == ROWS, "official row pins drift")
    need(
        snapshot["queensland_prep_entry_point_url"] == QCAA
        and snapshot["queensland_prep_alignment_url"] == ALIGN
        and QCAA in cross
        and ALIGN in cross,
        "QCAA pins drift",
    )
    need(
        set(snapshot["skill_practice_partial_codes"]) == SKILLS
        and snapshot["fictional_concept_rehearsal_only_codes"] == []
        and set(snapshot["knowledge_codes_held_for_authentic_local_sources"]) == HELD,
        "skill/knowledge boundary drift",
    )
    records = {
        r["code"]: r
        for r in imported["records"]
        if r.get("record_type") == "content_description" and r.get("code") in ROWS
    }
    need(set(records) == set(ROWS), "official Foundation HASS set absent")
    for code, number in ROWS.items():
        r = records[code]
        a = r["attributes"]
        need(
            r["source_row"] == number
            and a["learning_area"] == "Humanities and Social Sciences"
            and a["subject"] == "HASS F-6"
            and a["level"] == "Foundation Year",
            f"official row drift {code}",
        )
        exact = rf"^\| {code} \| {number} \| Humanities and Social Sciences · HASS F-6 · Foundation Year \| {re.escape(r['plain_text'])} \|"
        need(re.search(exact, cross, re.MULTILINE), f"crosswalk text drift {code}")
        if code in HELD:
            line = next(x for x in cross.splitlines() if x.startswith(f"| {code} |"))
            need("hold" in line.lower(), f"authentic hold missing {code}")
    for phrase in (
        "other states",
        "country/place",
        "not secure exams",
        "held for authentic local sources",
    ):
        need(phrase in cross.lower(), f"curriculum boundary missing: {phrase}")
    gate = read("LOCAL-SOURCE-INSERT.md").lower()
    need(
        "cultural authority" in gate and "hold the relevant knowledge claim" in gate,
        "local authority gate missing",
    )


def pedagogy():
    lessons = read("LESSONS.md")
    marks = list(re.finditer(r"^## Week \d+ · Day (\d+)\b", lessons, re.MULTILINE))
    need(
        [int(m.group(1)) for m in marks] == list(range(91, 101)),
        "ten-day lesson sequence",
    )
    for i, mark in enumerate(marks):
        day = int(mark.group(1))
        block = lessons[
            mark.end() : marks[i + 1].start() if i + 1 < len(marks) else len(lessons)
        ]
        times = [
            int(v)
            for v in re.findall(
                r"^\d+\. \*\*[^\n]*? · (\d+) min\.\*\*", block, re.MULTILINE
            )
        ]
        need(times == [3, 4, 5, 7, 4, 2], f"Day {day} timed script {times}")
        target = re.search(r"^\*\*Target/codes:\*\* ([^\n]+)", block, re.MULTILINE)
        need(target is not None, f"Day {day} target absent")
        codes = set(re.findall(r"\bAC9HSF(?:K|S)\d\d\b", target.group(1)))
        need(codes == EXPECTED[day], f"Day {day} code drift {codes}")
        need(
            "**Check/respond · 4 min.**" in block and "**Next move:**" in block,
            f"Day {day} feedback absent",
        )
        routes = table_row("LEARNER-CARDS.md", day, 3)
        need(len(set(routes)) == 3, f"Day {day} routes repeated")
        swaps = table_row("PRACTICE-SWAPS.md", day, 2)
        need(all("→" in s for s in swaps), f"Day {day} swaps not worked")
        table_row("FAMILY-CARER-OPTIONS.md", day, 1)
    need(
        "not permanent learning-style categories" in read("LEARNER-CARDS.md").lower()
        and "no child must disclose" in read("README.md").lower(),
        "access/privacy boundary missing",
    )


def checks():
    learner = read("STUDENT-CHECKS.md")
    key = read("teacher/KEY-AND-NEXT.md")
    need(
        re.findall(r"^## Check ([AB]) · Day (\d+)\b", learner, re.MULTILINE)
        == [("A", "95"), ("B", "100")],
        "fresh check schedule",
    )
    need(
        learner.count("\n\n1. ") == 2 and "KEY-AND-NEXT" not in learner,
        "public check/key separation",
    )
    for phrase in (
        "Pocket Card worked interpretation",
        "Shelf Tag worked interpretation",
        "fictional source-reading response",
    ):
        need(phrase.lower() in key.lower(), f"key boundary absent: {phrase}")
    ordinary = read("PRACTICE-SWAPS.md") + "\n".join(
        x
        for x in read("LEARNER-CARDS.md").splitlines()
        if not x.startswith(("| 95 |", "| 100 |"))
    )
    for name in ("Kira", "Nox", "Una", "Remy", "Ellis"):
        need(name not in ordinary, f"fresh check leak: {name}")
    need(
        "Does a later date prove" in learner
        and "Does that date order itself prove" in learner,
        "new-case unknown absent",
    )


def assets():
    alt = read("print/TEXT-ALTERNATIVES.md")
    tree = ast.parse(read("print/generate_print.py"))
    pages = next(
        node.value
        for node in tree.body
        if isinstance(node, ast.Assign)
        and any(isinstance(t, ast.Name) and t.id == "PAGES" for t in node.targets)
    )
    need(
        isinstance(pages, ast.Dict) and {k.value for k in pages.keys} == set(STEMS),
        "generator page set drift",
    )
    ns = {"s": "http://www.w3.org/2000/svg"}
    compact_alt = " ".join(alt.lower().split())
    for stem in STEMS:
        svg = ROOT / "print" / f"{stem}.svg"
        pdf = ROOT / "print" / f"{stem}.pdf"
        top = ET.parse(svg).getroot()
        need(
            top.get("width") == "210mm"
            and top.get("height") == "297mm"
            and top.get("viewBox") == "0 0 794 1123"
            and top.find("s:title", ns) is not None
            and top.find("s:desc", ns) is not None,
            f"{stem} SVG A4 or metadata",
        )
        need(
            len(top.findall(".//s:path", ns))
            + len(top.findall(".//s:circle", ns))
            + len(top.findall(".//s:polygon", ns))
            >= 2,
            f"{stem} not illustrated",
        )
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        raw = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        fonts = subprocess.check_output(["pdffonts", str(pdf)], text=True)
        compact_pdf = " ".join(raw.lower().split())
        need(
            re.search(r"Pages:\s+1\b", info) and "(A4)" in info and "DejaVu" in fonts,
            f"{stem} PDF A4/font drift",
        )
        need(
            len(raw.split()) >= 20 and "CC BY 4.0" in raw,
            f"{stem} accessible text/imprint drift",
        )
        need(
            f"{stem}.svg" in alt
            and f"{stem}.pdf" in alt
            and "tactile/no-print" in alt.lower(),
            f"{stem} alternative absent",
        )
        for phrase in PHRASES[stem]:
            need(
                phrase.lower() in compact_pdf and phrase.lower() in compact_alt,
                f"{stem} printed/alternative phrase mismatch {phrase}",
            )
    need(
        "bitstream" in read("print/DEJAVU-FONT-LICENSE.txt").lower(),
        "font licence absent",
    )
    a = ET.parse(ROOT / "print/check-a-pocket-card.svg").getroot()
    b = ET.parse(ROOT / "print/check-b-shelf-tag.svg").getroot()
    need(
        len(a.findall(".//s:circle", ns)) >= 2
        and len(b.findall(".//s:polygon", ns)) >= 1,
        "check shapes drift",
    )


def slug(s):
    s = re.sub(r"\[[^]]+\]\([^)]+\)", "", s)
    s = re.sub(r"[*_]", "", s).strip().lower()
    s = re.sub(r"[^\w -]", "", s)
    return s.replace(" ", "-")


def links(bootstrap):
    total = 0
    for page in ROOT.rglob("*.md"):
        for link in re.findall(
            r"(?<!!)\[[^]]+\]\(([^)]+)\)", page.read_text(encoding="utf-8")
        ):
            if link.startswith(("https://", "http://", "mailto:")):
                continue
            name, marker, anchor = unquote(link).partition("#")
            target = (page.parent / name).resolve() if name else page
            if bootstrap and target == ROOT / "manifest.json":
                total += 1
                continue
            need(
                target.is_file(),
                f"broken local link {page.relative_to(ROOT)} -> {link}",
            )
            if marker and target.suffix == ".md":
                heads = re.findall(
                    r"^#{1,6} (.+)$", target.read_text(encoding="utf-8"), re.MULTILINE
                )
                need(anchor in {slug(h) for h in heads}, f"broken anchor {link}")
            total += 1
    return total


def manifest_data():
    files = sorted(
        p
        for p in ROOT.rglob("*")
        if p.is_file()
        and p.name != "manifest.json"
        and "__pycache__" not in p.parts
        and ".ruff_cache" not in p.parts
    )
    names = {p.relative_to(ROOT).as_posix() for p in files}
    need(
        names == REQUIRED,
        f"pack file set drift missing={sorted(REQUIRED - names)} extra={sorted(names - REQUIRED)}",
    )
    for file in files:
        if file.suffix in (".md", ".py", ".svg", ".txt", ".json"):
            body = file.read_text(encoding="utf-8")
            need(
                body.endswith("\n")
                and "\r" not in body
                and all(line == line.rstrip() for line in body.splitlines()),
                f"{file.name} whitespace",
            )
    return {
        "schema": "subjectnest-authored-pack-manifest-v1",
        "pack": "foundation/hass/term-2/weeks-19-20",
        "created_at": "2026-09-30",
        "review_status": "author_desk_checked_pending_school_material_and_child_review",
        "curriculum_source_sha256": SHA,
        "curriculum_codes_partial_conditional": sorted(SKILLS),
        "curriculum_codes_fictional_rehearsal_only": [],
        "curriculum_codes_held_for_authentic_local_sources": sorted(HELD),
        "files": {
            str(p.relative_to(ROOT)): {
                "sha256": hashlib.sha256(p.read_bytes()).hexdigest(),
                "bytes": p.stat().st_size,
            }
            for p in files
        },
    }


def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    sources()
    pedagogy()
    checks()
    assets()
    n = links(args.write_manifest)
    expected = manifest_data()
    target = ROOT / "manifest.json"
    if args.write_manifest:
        target.write_text(
            json.dumps(expected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
        )
    else:
        need(
            json.loads(target.read_text(encoding="utf-8")) == expected,
            "SHA-256 manifest missing or stale",
        )
    print(
        f"PASS: nine exact ACARA rows/QCAA Prep; ten 25-minute scripts; 30 routes; 20 swaps; 10 bridges; two separate checks; eight A4 pairs; {n} local links; {len(expected['files'])} SHA-256 files"
    )


if __name__ == "__main__":
    main()
