#!/usr/bin/env python3
"""Fail-closed local audit for isolated Year 11 Queensland Chemistry starter."""
from __future__ import annotations

import argparse
import hashlib
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

PACK = Path(__file__).resolve().parent
QCAA = 'https://www.qcaa.qld.edu.au/downloads/senior-qce/syllabuses/snr_chemistry_25_syll.pdf'
STEMS = ('atom-model', 'count-mat', 'isotope-ion-fork')
TITLES = ('Atom model and its limits', 'A, Z and charge count mat',
          'Isotope or ion decision fork')
REQUIRED = {
    'README.md', 'LESSONS.md', 'MODEL-CARDS.md', 'LEARNER.md',
    'DAILY-CHOICES.md', 'DAILY-EXTRAS.md', 'STUDENT-CHECKS.md',
    'teacher/ANSWER-AND-NEXT.md', 'CURRICULUM-CROSSWALK.md',
    'SOURCE-AND-RIGHTS.md', 'QA-RUN-THROUGH.md', 'verify_pack.py',
    'print/generate_print.py', 'print/TEXT-ALTERNATIVES.md',
    'print/FONT-RIGHTS.md', 'print/dejavu-font-copyright.txt',
}
for stem in STEMS:
    REQUIRED.update((f'print/{stem}.svg', f'print/{stem}.pdf'))


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def get(rel: str) -> str:
    return (PACK / rel).read_text(encoding='utf-8')


def curriculum_and_rights() -> None:
    readme, cross, source = (get(rel) for rel in
                             ('README.md', 'CURRICULUM-CROSSWALK.md',
                              'SOURCE-AND-RIGHTS.md'))
    for body in (readme, cross, source):
        need(QCAA in body and '2025 v1.3' in body,
             'QCAA Chemistry official URL/version absent')
    need('29 September 2026' in cross and '29 September 2026' in source,
         'live source check date absent')
    for marker in ('Unit 1', 'Topic 1', '20 hours', '55 hours', 'printed p.20',
                   'printed p.21', 'printed p.22', 'Atomic structure',
                   'bullet 1', 'bullet 2', 'bullet 3', 'Isotopes',
                   'not school instruments'):
        need(marker.lower() in cross.lower(), f'QCAA crosswalk/limit absent: {marker}')
    for marker in ('Aufbau', 'mass spectrometry', 'flame test', 'bonding',
                   'science inquiry', 'radioactivity', 'public'):
        need(marker.lower() in (readme + cross).lower(),
             f'uncovered scope/public boundary missing: {marker}')
    need('AC9CH11' not in cross and 'AC9' not in readme,
         'invented national senior curriculum code risk')
    need('no chemicals' in readme.lower() and 'no classroom pilot' in readme.lower(),
         'safety or pilot boundary missing')
    for marker in ('IUPAC', 'CIAAW', 'NIST', 'DejaVu', 'CC BY 4.0'):
        need(marker in source, f'rights/source source missing: {marker}')


def daily_structure() -> None:
    lesson, learner, choices, extras = (get(rel) for rel in
                                       ('LESSONS.md', 'LEARNER.md',
                                        'DAILY-CHOICES.md', 'DAILY-EXTRAS.md'))
    for day in range(1, 11):
        parts = re.findall(rf'^### Day {day} ·(.+?)(?=^### Day |\Z)', lesson,
                           flags=re.MULTILINE | re.DOTALL)
        need(len(parts) == 1 and '**Target:**' in parts[0] and
             '**Prepare:**' in parts[0], f'Day {day}: script/prepare missing or repeated')
        stages = (('Launch', 'Source access', 'Independent plan',
                   'Independent response', 'Self-audit', 'Submit') if day in (5, 10)
                  else ('Launch', 'Model', 'Guided reading', 'Practice route',
                        'Audit', 'Exit'))
        timings = []
        for stage in stages:
            found = re.findall(rf'\*\*{re.escape(stage)} · (\d+) min\.\*\*', parts[0])
            need(len(found) == 1, f'Day {day}: stage {stage} missing or repeated')
            timings.append(int(found[0]))
        need(timings == [2, 4, 5, 7, 4, 3],
             f'Day {day}: timing drift {timings}')
        for kind, body, cols in (('learner', learner, 3), ('routes', choices, 4),
                                 ('extra', extras, 1)):
            hits = re.findall(rf'^\| {day} (?:·|\|)(.+)$', body,
                              flags=re.MULTILINE)
            need(len(hits) == 1, f'Day {day}: {kind} row missing/repeated')
            cells = [c.strip() for c in hits[0].strip().strip('|').split('|')]
            need(len(cells) == cols and all(cells),
                 f'Day {day}: {kind} row incomplete')
            if kind == 'routes':
                need(len(set(cells[:3])) == 3,
                     f'Day {day}: duplicated access routes')
    need('independent reading remains a separate construct' in
         get('QA-RUN-THROUGH.md'), 'access/construct distinction missing')


def _cells(row: str) -> list[str]:
    return [c.strip() for c in row.strip().strip('|').split('|')]


def _tail_int(cell: str) -> int:
    match = re.search(r'(?:=)?([+-]?\d+)$', cell.replace('−', '-'))
    need(match is not None, f'cannot parse final integer from {cell!r}')
    return int(match.group(1))


def arithmetic_and_checks() -> None:
    practice, check, key = (get(rel) for rel in
                            ('MODEL-CARDS.md', 'STUDENT-CHECKS.md',
                             'teacher/ANSWER-AND-NEXT.md'))
    for label in 'ABCDEF':
        need(len(re.findall(rf'^## Card {label} ·', practice, flags=re.MULTILINE)) == 1,
             f'practice Card {label} absent/repeated')
    for label in 'RSTU':
        need(len(re.findall(rf'^\*\*File {label} ·', check, flags=re.MULTILINE)) == 1,
             f'fresh File {label} absent/repeated')
    for unique in ('fictional ceramics archive card', 'fictional transit-sign card',
                   'fictional conservation-card label'):
        need(unique in check, f'held-out source absent: {unique}')
        for rel in ('LESSONS.md', 'MODEL-CARDS.md', 'LEARNER.md',
                    'DAILY-CHOICES.md'):
            need(unique not in get(rel), f'held-out source leaked to {rel}')
    for source_fact in (
        'neutral boron-10 atom', '`^{10}_{5}B`', 'Z 5 and mass number A 10',
        'neutral boron-11 atom', '`^{11}_{5}B`', 'Z 5 and mass number A 11',
        'calcium-40 **2+ ion**', '`^{40}_{20}Ca^{2+}`', 'Z 20, A 40',
        'fluorine-19 **1− ion**', '`^{19}_{9}F^{−}`', 'Z 9, A 19',
    ):
        need(source_fact in check, f'check source datum/status drift: {source_fact}')
    need(len(re.findall(r'^\| (?:[1-9]|10) \|', key, flags=re.MULTILINE)) == 10,
         'ten daily answer/next-move rows required')
    need('publicly accessible' in check and 'public by URL' in key,
         'public check/key status absent')
    need('teacher/ANSWER-AND-NEXT.md' not in get('LEARNER.md'),
         'clean learner page directly links answer key')

    cases = {
        'R · boron-10': (5, 10, 0),
        'S · boron-11': (5, 11, 0),
        'T · calcium-40 2+': (20, 40, 2),
        'U · fluorine-19 1−': (9, 19, -1),
    }
    for name, (z, a, q) in cases.items():
        found = re.findall(rf'^\| {re.escape(name)} \|(.+)$', key,
                           flags=re.MULTILINE)
        need(len(found) == 1, f'check key row missing/repeated: {name}')
        cells = _cells(found[0])
        need(len(cells) == (6 if name[0] in 'RS' else 7),
             f'check key cell count drift: {name}')
        # The first column is Z because the row name has already been consumed.
        given_z, given_a, listed_p = map(_tail_int, cells[:3])
        listed_n = _tail_int(cells[3])
        listed_e = _tail_int(cells[4])
        need((given_z, given_a, listed_p, listed_n, listed_e) ==
             (z, a, z, a - z, z - q),
             f'check key arithmetic differs from independently computed Z/A/q: {name}')
        need(z + (a - z) == a and z - (z - q) == q,
             f'case invariant failed: {name}')
        if name[0] in 'TU':
            need(_tail_int(cells[5]) == q and _tail_int(cells[6]) == z,
                 f'ion charge or same-isotope neutral e check fails: {name}')
        else:
            need(_tail_int(cells[5]) == a,
                 f'isotope reverse-addition check fails: {name}')
    # Independently recomputed selected practice examples must appear in the key.
    for symbol, a, z, q, snippet in (
        ('C', 12, 6, 0, 'C-12: p6 n6'),
        ('O', 16, 8, 0, 'O-16: p8 n8'),
        ('Mg', 24, 12, 2, 'Mg-24 2+ p12 n12 e10'),
        ('Cl', 35, 17, -1, 'Cl-35 1− p17 n18 e18'),
    ):
        need((a - z) >= 0 and (z - q) >= 0, f'bad model constants {symbol}')
        need(snippet in key, f'practice answer drift: {symbol}-{a}, q={q}')
    for practice_fact in ('carbon-12, Z 6, **neutral** | p 6, n 6, e 6',
                          'carbon-13, Z 6, **neutral** | p 6, n 7, e 6',
                          'magnesium-24, Z 12, **2+ ion** | p 12, n 12, e 10',
                          'chlorine-35, Z 17, **1− ion** | p 17, n 18, e 18'):
        need(practice_fact in practice,
             f'practice source/count drift: {practice_fact}')
    for marker in ('same Z', 'neutron count', 'different elements',
                   'reaction occurred', 'cannot prove', 'next move'):
        need(marker.lower() in key.lower(),
             f'worked reasoning or feedback boundary absent: {marker}')


def slug(heading: str) -> str:
    heading = re.sub(r'<[^>]+>', '', heading.lower())
    heading = re.sub(r'[^\w\- ]', '', heading)
    return heading.replace(' ', '-')


def local_links() -> int:
    count = 0
    for md in PACK.rglob('*.md'):
        for url in re.findall(r'\[[^]]+\]\(([^)]+)\)', md.read_text(encoding='utf-8')):
            if url.startswith(('https://', 'http://', 'mailto:')):
                continue
            base, _, anchor = unquote(url).partition('#')
            dest = (md.parent / base).resolve() if base else md
            need(dest.exists() and dest.is_relative_to(PACK),
                 f'broken/escaping link {md.relative_to(PACK)} -> {url}')
            if anchor and dest.suffix.lower() == '.md':
                headings = re.findall(r'^#{1,6} (.+)$', dest.read_text(encoding='utf-8'),
                                      flags=re.MULTILINE)
                need(anchor in {slug(h) for h in headings},
                     f'broken anchor {md.relative_to(PACK)} -> {url}')
            count += 1
    return count


def print_aids() -> None:
    alternatives = get('print/TEXT-ALTERNATIVES.md')
    rights = get('print/FONT-RIGHTS.md')
    need(alternatives.count('Tactile route:') == 3,
         'one complete tactile route per aid required')
    need('untagged' in rights and 'DejaVu' in rights,
         'PDF and font accessibility boundary missing')
    for stem, title in zip(STEMS, TITLES, strict=True):
        svg = PACK / 'print' / f'{stem}.svg'
        pdf = PACK / 'print' / f'{stem}.pdf'
        root = ET.parse(svg).getroot()
        need((root.attrib.get('width'), root.attrib.get('height'),
              root.attrib.get('viewBox')) ==
             ('210mm', '297mm', '0 0 794 1123'), f'{stem}: SVG A4 drift')
        labels = [child.text or '' for child in root]
        need(title in labels and any(len(label) > 100 for label in labels),
             f'{stem}: SVG title/description missing')
        info = subprocess.run(['pdfinfo', str(pdf)], text=True, capture_output=True,
                              check=True).stdout
        need(re.search(r'^Pages:\s+1$', info, flags=re.MULTILINE) is not None and
             re.search(r'^Page size:\s+595\.\d+ x 841\.\d+ pts \(A4\)',
                       info, flags=re.MULTILINE) is not None,
             f'{stem}: PDF not single-page A4')
        fonts = subprocess.run(['pdffonts', str(pdf)], text=True, capture_output=True,
                               check=True).stdout
        need('DejaVuSans' in fonts and ' yes ' in fonts,
             f'{stem}: embedded DejaVu absent')
        words = subprocess.run(['pdftotext', str(pdf), '-'], text=True,
                               capture_output=True, check=True).stdout
        need(title in words and len(words.split()) > 55,
             f'{stem}: selectable PDF text absent')
        need(f'{stem}.svg' in alternatives and f'{stem}.pdf' in alternatives,
             f'{stem}: full-text alternative links absent')


def receipt(write: bool) -> None:
    paths = sorted((p for p in PACK.rglob('*') if p.is_file() and
                    p.name != 'MANIFEST.sha256' and
                    not {'__pycache__', '.ruff_cache'}.intersection(p.parts)),
                   key=lambda p: p.relative_to(PACK).as_posix())
    actual = {p.relative_to(PACK).as_posix() for p in paths}
    need(actual == REQUIRED,
         f'file inventory drift: missing={REQUIRED - actual}; extra={actual - REQUIRED}')
    expected = '\n'.join(f'{hashlib.sha256(p.read_bytes()).hexdigest()}  '
                         f'{p.relative_to(PACK).as_posix()}' for p in paths) + '\n'
    manifest = PACK / 'MANIFEST.sha256'
    if write:
        manifest.write_text(expected, encoding='utf-8')
    else:
        need(manifest.exists() and manifest.read_text(encoding='utf-8') == expected,
             'SHA-256 receipt absent or stale')


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument('--write-manifest', action='store_true')
    args = parser.parse_args()
    curriculum_and_rights()
    daily_structure()
    arithmetic_and_checks()
    links = local_links()
    print_aids()
    receipt(args.write_manifest)
    print('PASS: QCAA Chemistry 2025 v1.3 partial Unit 1 Topic 1; ten timed Days 1–10; '
          'thirty equivalent-target routes; two new public checks with recomputed arithmetic; '
          f'{links} local links/anchors; three A4 SVG/PDF/full-text aids; SHA-256 receipt')


if __name__ == '__main__':
    main()
