#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Read-only structure, maths, curriculum, asset, link and manifest audit."""
from __future__ import annotations

import argparse
import hashlib
import json
import math
import re
import subprocess
import sys
import xml.etree.ElementTree as ET
from fractions import Fraction
from pathlib import Path
from urllib.parse import unquote

ROOT = Path(__file__).resolve().parent
ACARA = ROOT.parents[4] / "data/frameworks/acara-v9.json"
SOURCE_SHA = "db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3"
ROWS = {
    "AC9M6N02": (17823, "identify and describe the properties of prime, composite and square numbers and use these properties to solve problems and simplify calculations"),
    "AC9M6N03": (17829, "apply knowledge of equivalence to compare, order and represent common fractions including halves, thirds and quarters on the same number line and justify their order"),
}
REQUIRED = {
    "README.md", "LESSONS.md", "MATERIALS.md", "LEARNER.md", "DAILY-CHOICES.md",
    "DAILY-EXTRAS.md", "STUDENT-CHECKS.md", "teacher/ANSWER-AND-NEXT.md",
    "CURRICULUM-CROSSWALK.md", "SOURCES-AND-REVIEW.md", "RUN-THROUGH.md",
    "CODE-LICENSE.txt", "verify_pack.py", "print/generate_print.py",
    "print/TEXT-ALTERNATIVES.md", "print/FONT-RIGHTS.md",
    "print/dejavu-font-copyright.txt",
}
STEMS = ("factor-array-lab", "square-factor-strip", "fraction-lines", "equal-whole-bars")
for stem in STEMS:
    REQUIRED |= {f"print/{stem}.svg", f"print/{stem}.pdf"}


def require(ok: bool, detail: str) -> None:
    if not ok:
        raise AssertionError(detail)


def source_records() -> tuple[dict, dict[str, dict]]:
    data = json.loads(ACARA.read_text(encoding="utf-8"))
    require(data["source_sha256"] == SOURCE_SHA, "ACARA source workbook SHA drift")
    require(data["framework"] == "Australian Curriculum Version 9.0", "framework label drift")
    require(data["source_url"].startswith("https://www.australiancurriculum.edu.au/"), "official source URL drift")
    found = {r["code"]: r for r in data["records"] if r.get("record_type") == "content_description" and r.get("code") in ROWS}
    require(set(found) == set(ROWS), "Year 6 maths content records missing")
    for code, (row, wording) in ROWS.items():
        r = found[code]
        require((r["source_row"], r["plain_text"]) == (row, wording), f"{code}: row/wording drift")
        a = r["attributes"]
        require((a["level"], a["learning_area"], a["subject"]) == ("Year 6", "Mathematics", "Mathematics"),
                f"{code}: wrong level/area/subject")
    return data, found


def crosswalk(data: dict, found: dict[str, dict]) -> str:
    out = ["# Australian Curriculum v9 crosswalk · Year 6 mathematics Weeks 3–4", "",
           "These are **partial content-description links** for ten lessons, not the whole Year 6 mathematics curriculum, an achievement-standard judgement or state/territory approval. The source wording is pinned exactly, apart from plain-text display whitespace. Each taught slice below names the actual task and its limit.", "",
           f"Source: [official ACARA v9 workbook]({data['source_url']}), retrieved {data['retrieved_at']}; workbook SHA-256 `{data['source_sha256']}`. Row numbers allow audit against that snapshot.", "",
           "| Official code | Workbook row | Official level / area | Official content description | Taught slice and limit |",
           "| --- | ---: | --- | --- | --- |"]
    slices = {
        "AC9M6N02": "Days 11–15: factor pairs, prime/composite/square overlap, layout constraints and one regrouped product. This fortnight is a first encounter and does not prove broad mastery of number properties.",
        "AC9M6N03": "Days 16–20: halves, thirds, quarters and their equivalents on fixed-whole 0–1/0–2 lines, with exact order and reason. Wider fraction operations are in later weeks, not assessed here.",
    }
    for code in ROWS:
        r = found[code]
        cells = [code, str(r["source_row"]), "Year 6 / Mathematics", r["plain_text"], slices[code]]
        out.append("| " + " | ".join(c.replace("|", "\\|") for c in cells) + " |")
    out += ["",
            "© Australian Curriculum, Assessment and Reporting Authority (ACARA) 2010 to present, unless otherwise indicated. Downloaded from the Australian Curriculum website (accessed 29 September 2026) and modified for plain-text display. Curriculum material is licensed under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/), subject to [terms and exclusions](https://www.australiancurriculum.edu.au/copyright-and-terms-of-use). ACARA does not endorse SubjectNest; SubjectNest is not affiliated with, sponsored or approved by ACARA.", "",
            "This is a dated source snapshot, not live synchronisation. Check the current official curriculum and each jurisdiction's implementation before reissue. See [sources, rights and review](SOURCES-AND-REVIEW.md).", ""]
    return "\n".join(out)


def local_links() -> None:
    for md in ROOT.rglob("*.md"):
        content = md.read_text(encoding="utf-8")
        for raw in re.findall(r"\]\(([^)]+)\)", content):
            target = unquote(raw.split("#", 1)[0])
            if not target or target.startswith(("https://", "http://", "mailto:")):
                continue
            require((md.parent / target).is_file(), f"Broken local link in {md.relative_to(ROOT)}: {raw}")


def content_checks() -> None:
    for item in REQUIRED:
        require((ROOT / item).is_file(), f"Missing file: {item}")
    lesson = (ROOT / "LESSONS.md").read_text(encoding="utf-8")
    learner = (ROOT / "LEARNER.md").read_text(encoding="utf-8")
    choices = (ROOT / "DAILY-CHOICES.md").read_text(encoding="utf-8")
    extras = (ROOT / "DAILY-EXTRAS.md").read_text(encoding="utf-8")
    checks = (ROOT / "STUDENT-CHECKS.md").read_text(encoding="utf-8")
    key = (ROOT / "teacher/ANSWER-AND-NEXT.md").read_text(encoding="utf-8")
    for name, body, marker in (("lessons", lesson, r"^### Day (\d+) ·"),
                               ("learner", learner, r"^## Day (\d+) ·"),
                               ("routes", choices, r"^## Day (\d+) ·"),
                               ("extras", extras, r"^\| (\d+) \|")):
        require([int(x) for x in re.findall(marker, body, flags=re.MULTILINE)] == list(range(11, 21)),
                f"{name}: Days 11–20 not exactly once in order")
    choice_sections = re.split(r"^## Day \d+ ·", choices, flags=re.MULTILINE)[1:]
    for day, section in enumerate(choice_sections, 11):
        require(re.findall(r"^- \*\*([ABC]) ·", section, flags=re.MULTILINE) == ["A", "B", "C"],
                f"Day {day}: exactly three ordered routes required")
        for letter in "ABC":
            require(f"D{day}-{letter}" in key, f"D{day}-{letter}: staff answer missing")
    sections = re.split(r"^### Day \d+ ·", lesson, flags=re.MULTILINE)[1:]
    for day, section in enumerate(sections, 11):
        for stage, minutes in (("Launch", 2), ("Model", 5), ("Guided", 6), ("Note", 2)):
            require(f"**{stage} · {minutes} min.**" in section, f"Day {day}: {stage} timing")
        if day in (15, 20):
            require("**Independent check · 6 min.**" in section and "**Check exit · 4 min.**" in section,
                    f"Day {day}: fresh check timing")
            require("**Choice · 6 min.**" not in section, f"Day {day}: coached route replaces fresh check")
        else:
            require("**Choice · 6 min.**" in section and "**Exit · 4 min.**" in section,
                    f"Day {day}: choice/exit timing")
        require("**Home/extension:**" in section, f"Day {day}: missing home/extension")
        wanted = "AC9M6N02" if day <= 15 else "AC9M6N03"
        require(f"**Code:** `{wanted}`" in section, f"Day {day}: official code sequence mismatch")
    require(len(re.findall(r"^\d\. \*\*", checks, flags=re.MULTILINE)) == 7, "fresh checks: four plus three numbered items required")
    require("## Day 15 held-out check" in key and "## Day 20 held-out check" in key, "separate staff check key absent")
    pre15 = choices.split("## Day 15")[0] + extras.split("| 15 |")[0] + learner.split("## Day 15")[0]
    for marker in ("27", "37", "64", "31", "33"):
        require(not re.search(rf"\b{marker}\b", pre15), f"Day15 held-out number leaked into earlier learner practice: {marker}")
    pre20 = choices.split("## Day 20")[0] + extras.split("| 20 |")[0] + learner.split("## Day 20")[0]
    for marker in ("5/3", "7/4"):
        require(marker not in pre20, f"Day20 held-out fraction leaked into earlier learner practice: {marker}")
    require("Later practice" in choices and "the fresh check" in extras, "check/practice boundary unclear")
    local_links()


def factor_pairs(n: int) -> list[tuple[int, int]]:
    return [(a, n // a) for a in range(1, math.isqrt(n) + 1) if n % a == 0]


def mathematics_checks() -> None:
    expected = {
        12: [(1, 12), (2, 6), (3, 4)],
        18: [(1, 18), (2, 9), (3, 6)],
        20: [(1, 20), (2, 10), (4, 5)],
        21: [(1, 21), (3, 7)],
        23: [(1, 23)],
        27: [(1, 27), (3, 9)],
        31: [(1, 31)],
        33: [(1, 33), (3, 11)],
        37: [(1, 37)],
        64: [(1, 64), (2, 32), (4, 16), (8, 8)],
    }
    for n, pairs in expected.items():
        require(factor_pairs(n) == pairs, f"factor pair audit failed for {n}")
    require(12 * 15 == 180 and 8 * 25 == 200, "product audit")
    marks = {"quarter": Fraction(1, 4), "third": Fraction(1, 3),
             "half": Fraction(1, 2), "two-thirds": Fraction(2, 3),
             "three-quarters": Fraction(3, 4), "five-quarters": Fraction(5, 4),
             "three-halves": Fraction(3, 2), "five-thirds": Fraction(5, 3),
             "seven-quarters": Fraction(7, 4)}
    expected_ticks = {"quarter": 3, "third": 4, "half": 6, "two-thirds": 8,
                      "three-quarters": 9, "five-quarters": 15, "three-halves": 18,
                      "five-thirds": 20, "seven-quarters": 21}
    require({name: value * 12 for name, value in marks.items()} == expected_ticks, "fraction tick audit")
    require(Fraction(7, 4) - Fraction(5, 3) == Fraction(1, 12), "Day20 gap audit")
    require(Fraction(3, 4) - Fraction(2, 3) == Fraction(1, 12), "Day17 gap audit")
    require(Fraction(3, 2) - Fraction(5, 4) == Fraction(1, 4), "Day18 gap audit")
    require((24 // 3, 24 // 2, 24 * 3 // 4, 12 // 2) == (8, 12, 18, 6), "tile whole/count audit")
    key = (ROOT / "teacher/ANSWER-AND-NEXT.md").read_text(encoding="utf-8")
    for fragment in ("27 distinct positive pairs", "37 is prime", "64=`8×8`", "33 works as `3×11`",
                     "`5/3=20/12`", "`7/4=21/12`", "`24÷3=8`", "`12÷2=6`"):
        require(fragment in key, f"staff arithmetic key drift: {fragment}")


def print_checks() -> None:
    subprocess.run([sys.executable, str(ROOT / "print/generate_print.py")], check=True, capture_output=True, text=True)
    ns = "{http://www.w3.org/2000/svg}"
    for stem in STEMS:
        svg = ROOT / f"print/{stem}.svg"
        pdf = ROOT / f"print/{stem}.pdf"
        tree = ET.parse(svg).getroot()
        require((tree.attrib.get("width"), tree.attrib.get("height"), tree.attrib.get("viewBox")) ==
                ("210mm", "297mm", "0 0 210 297"), f"{stem}: SVG not A4")
        require(tree.find(ns + "title") is not None and tree.find(ns + "desc") is not None,
                f"{stem}: SVG title/description missing")
        info = subprocess.check_output(["pdfinfo", str(pdf)], text=True)
        require("Pages:           1" in info and "Page size:       595.276 x 841.89 pts (A4)" in info,
                f"{stem}: PDF not one-page A4")
        copy = subprocess.check_output(["pdftotext", str(pdf), "-"], text=True)
        require(len(copy.strip()) > 150 and "SubjectNest" in copy, f"{stem}: PDF selectable text absent")
        fonts = subprocess.check_output(["pdffonts", str(pdf)], text=True)
        require("DejaVu" in fonts, f"{stem}: PDF font drift")
    lines = subprocess.check_output(["pdftotext", str(ROOT / "print/fraction-lines.pdf"), "-"], text=True)
    require("12 equal gaps" in lines and "24 in all" in lines and "DIFFERENT physical scales" in lines,
            "fraction line scale labels absent")


def manifest_content() -> str:
    rows = []
    for path in sorted(ROOT.rglob("*")):
        if path.is_file() and path.name != "MANIFEST.sha256" and "__pycache__" not in path.parts:
            rows.append(f"{hashlib.sha256(path.read_bytes()).hexdigest()}  {path.relative_to(ROOT).as_posix()}")
    return "\n".join(rows) + "\n"


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--write-crosswalk", action="store_true")
    parser.add_argument("--write-manifest", action="store_true")
    args = parser.parse_args()
    data, found = source_records()
    crosswalk_file = ROOT / "CURRICULUM-CROSSWALK.md"
    wanted = crosswalk(data, found)
    if args.write_crosswalk:
        crosswalk_file.write_text(wanted, encoding="utf-8")
    else:
        require(crosswalk_file.is_file() and crosswalk_file.read_text(encoding="utf-8") == wanted,
                "exact curriculum crosswalk missing/stale")
    content_checks()
    mathematics_checks()
    print_checks()
    manifest_file = ROOT / "MANIFEST.sha256"
    hashes = manifest_content()
    if args.write_manifest:
        manifest_file.write_text(hashes, encoding="utf-8")
    else:
        require(manifest_file.is_file() and manifest_file.read_text(encoding="utf-8") == hashes,
                "hash manifest missing/stale")
    print("PASS: two exact Year 6 ACARA maths rows; 10 x 25-minute lessons; 30 routes; "
          "7 fresh items; factor/fraction arithmetic; four accessible A4 aid pairs; links; hashes")


if __name__ == "__main__":
    try:
        main()
    except (AssertionError, subprocess.CalledProcessError, KeyError, ValueError) as exc:
        print(f"FAIL: {exc}", file=sys.stderr)
        raise SystemExit(1)
