#!/usr/bin/env python3
"""Fail-closed audit of isolated Queensland Year 11 Business starter."""
from __future__ import annotations

import argparse
import hashlib
import re
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import unquote

PACK = Path(__file__).resolve().parent
QCAA = 'https://www.qcaa.qld.edu.au/downloads/senior-qce/syllabuses/snr_business_25_syll.pdf'
STEMS = ('evidence-ladder', 'lifecycle-goals', 'environment-swot',
         'functions-criteria')
TITLES = ('Evidence ladder', 'Life cycle and goals', 'Environment to SWOT',
          'Functions and criteria')
REQUIRED = {
    'README.md', 'CASE-CARDS.md', 'LESSONS.md', 'LEARNER.md',
    'DAILY-CHOICES.md', 'DAILY-EXTRAS.md', 'STUDENT-CHECKS.md',
    'teacher/ANSWER-AND-NEXT.md', 'CURRICULUM-CROSSWALK.md',
    'SOURCE-AND-RIGHTS.md', 'QA-RUN-THROUGH.md', 'verify_pack.py',
    'print/generate_print.py', 'print/TEXT-ALTERNATIVES.md',
    'print/FONT-RIGHTS.md', 'print/dejavu-font-copyright.txt',
}
for stem in STEMS:
    REQUIRED.update((f'print/{stem}.svg', f'print/{stem}.pdf'))


def need(ok: bool, message: str) -> None:
    if not ok:
        raise AssertionError(message)


def get(rel: str) -> str:
    return (PACK / rel).read_text(encoding='utf-8')


def curriculum_and_rights() -> None:
    readme, cross, source = (get(rel) for rel in
                             ('README.md', 'CURRICULUM-CROSSWALK.md',
                              'SOURCE-AND-RIGHTS.md'))
    for body in (readme, cross, source):
        need(QCAA in body and '2025 v1.3' in body,
             'QCAA Business URL/version absent')
    need('29 September 2026' in cross and '29 September 2026' in source,
         'live source-check date missing')
    for marker in ('Unit 1', 'Business creation', 'Topic 1',
                   'Fundamentals of business', 'printed p.14', 'printed p.15',
                   'printed p.13', 'printed p.5', '55-hour', 'SWOT',
                   'authentic business case studies', 'not school instruments'):
        need(marker.lower() in cross.lower(),
             f'official point/hold absent: {marker}')
    need('AC9BUS11' not in cross and 'AC9' not in readme,
         'invented national senior code risk')
    for marker in ('fictional', 'not financial', 'no classroom pilot',
                   'authentic case studies'):
        need(marker in readme.lower(), f'claim/scope boundary absent: {marker}')
    for marker in ('QCAA', 'DejaVu', 'CC BY 4.0', 'No actual company',
                   'no classroom pilot'):
        need(marker.lower() in source.lower(),
             f'rights/provenance boundary absent: {marker}')


def daily_structure() -> None:
    lesson, learner, choices, extras = (get(rel) for rel in
                                       ('LESSONS.md', 'LEARNER.md',
                                        'DAILY-CHOICES.md', 'DAILY-EXTRAS.md'))
    for day in range(1, 11):
        scripts = re.findall(rf'^### Day {day} ·(.+?)(?=^### Day |\Z)', lesson,
                             flags=re.MULTILINE | re.DOTALL)
        need(len(scripts) == 1 and '**Target:**' in scripts[0] and
             '**Prepare:**' in scripts[0],
             f'Day {day}: target/preparation script missing/repeated')
        stages = (('Launch', 'Source access', 'Independent plan',
                   'Independent response', 'Self-audit', 'Submit') if day in (5, 10)
                  else ('Launch', 'Model', 'Guided reading', 'Practice route',
                        'Audit', 'Exit'))
        timings = []
        for stage in stages:
            match = re.findall(rf'\*\*{re.escape(stage)} · (\d+) min\.\*\*', scripts[0])
            need(len(match) == 1, f'Day {day}: timed {stage} missing/repeated')
            timings.append(int(match[0]))
        need(timings == [2, 4, 5, 7, 4, 3],
             f'Day {day}: timing drift {timings}')
        for kind, body, cols in (('learner', learner, 3), ('routes', choices, 4),
                                 ('extra', extras, 1)):
            rows = re.findall(rf'^\| {day} (?:·|\|)(.+)$', body,
                              flags=re.MULTILINE)
            need(len(rows) == 1, f'Day {day}: {kind} row missing/repeated')
            cells = [cell.strip() for cell in rows[0].strip().strip('|').split('|')]
            need(len(cells) == cols and all(cells),
                 f'Day {day}: {kind} row incomplete')
            if kind == 'routes':
                need(len(set(cells[:3])) == 3,
                     f'Day {day}: routes duplicate')
    need('reading ability considered separately' in
         get('QA-RUN-THROUGH.md'),
         'access and reading construct not separated')


def cases_checks_and_arithmetic() -> None:
    cases, checks, key = (get(rel) for rel in
                          ('CASE-CARDS.md', 'STUDENT-CHECKS.md',
                           'teacher/ANSWER-AND-NEXT.md'))
    for label in 'ABCDEF':
        need(len(re.findall(rf'^## Card {label} ·', cases, flags=re.MULTILINE)) == 1,
             f'practice Card {label} absent/repeated')
    for label in 'GH':
        need(len(re.findall(rf'^\*\*File {label} —', checks, flags=re.MULTILINE)) == 1,
             f'held-out File {label} absent/repeated')
    for unique in ('Borrow Bench', 'Page Pop'):
        need(unique in checks, f'fresh case absent: {unique}')
        for rel in ('CASE-CARDS.md', 'LEARNER.md', 'DAILY-CHOICES.md'):
            need(unique not in get(rel), f'held-out case leaked to {rel}')
    need(len(re.findall(r'^\| (?:[1-9]|10) \|', key, flags=re.MULTILINE)) == 10,
         'ten worked daily next-move rows required')
    need('publicly accessible' in checks and 'public by URL' in key,
         'public check/key status absent')
    need('teacher/ANSWER-AND-NEXT.md' not in get('LEARNER.md'),
         'clean learner page directly links key')

    for datum in ('**40 orders**', '**34 were finished on time**',
                  '**36 of 40 on time**', '**20 voluntary model feedback slips**',
                  '**12 slips were positive**', '**18 positive of 20**',
                  '**$80**', '**16 positive of 20**', '**$30**',
                  '**15 positive of 20**'):
        need(datum in cases, f'practice datum drift: {datum}')
    for datum in ('**two Saturdays next month**', '**one Saturday next month only**',
                  '**three public sessions next month**', '**23 of 30**',
                  '**at least 27 of 30**', '**$200**', '**28 of 30**',
                  '**$90**', '**26 of 30**'):
        need(datum in checks, f'fresh check datum drift: {datum}')
    need(36 - 34 == 2 and 16 - 12 == 4 and 15 - 12 == 3 and
         80 // 4 == 20 and 30 // 3 == 10 and 3 - 2 == 1 and
         3 - 1 == 2 and 28 - 23 == 5 and 26 - 23 == 3 and
         200 // 5 == 40 and 90 // 3 == 30,
         'structured case arithmetic invariant failed')
    for worked in ('36−34=**2 on-time orders**', '$80÷4=**$20**',
                   '$30÷3=**$10**', '3−2=1', '3−1=2',
                   '28−23=**5**', '26−23=**3**',
                   '$200÷5=**$40**', '$90÷3=**$30**'):
        need(worked in key, f'worked key arithmetic drift: {worked}')
    plain = re.sub(r'[*_`]', '', key + checks).lower()
    for boundary in ('not a booking', 'unverified', 'overall efficiency',
                     'school assessment', 'next move', 'permission'):
        need(boundary.lower() in plain,
             f'case reasoning/safety limit absent: {boundary}')


def slug(value: str) -> str:
    value = re.sub(r'<[^>]+>', '', value.lower())
    value = re.sub(r'[^\w\- ]', '', value)
    return value.replace(' ', '-')


def local_links() -> int:
    count = 0
    for md in PACK.rglob('*.md'):
        for url in re.findall(r'\[[^]]+\]\(([^)]+)\)', md.read_text(encoding='utf-8')):
            if url.startswith(('https://', 'http://', 'mailto:')):
                continue
            base, _, anchor = unquote(url).partition('#')
            dest = (md.parent / base).resolve() if base else md
            need(dest.exists() and dest.is_relative_to(PACK),
                 f'broken or escaping local link {md.relative_to(PACK)} -> {url}')
            if anchor and dest.suffix.lower() == '.md':
                headings = re.findall(r'^#{1,6} (.+)$', dest.read_text(encoding='utf-8'),
                                      flags=re.MULTILINE)
                need(anchor in {slug(h) for h in headings},
                     f'broken anchor {md.relative_to(PACK)} -> {url}')
            count += 1
    return count


def print_aids() -> None:
    alternatives, rights = (get(rel) for rel in
                            ('print/TEXT-ALTERNATIVES.md',
                             'print/FONT-RIGHTS.md'))
    need(alternatives.count('Tactile route:') == 4,
         'four exact full-word tactile routes required')
    need('untagged' in rights and 'DejaVu' in rights,
         'font/PDF access boundary missing')
    for stem, title in zip(STEMS, TITLES, strict=True):
        svg = PACK / 'print' / f'{stem}.svg'
        pdf = PACK / 'print' / f'{stem}.pdf'
        root = ET.parse(svg).getroot()
        need((root.attrib.get('width'), root.attrib.get('height'),
              root.attrib.get('viewBox')) ==
             ('210mm', '297mm', '0 0 794 1123'),
             f'{stem}: SVG A4 geometry drift')
        labels = [child.text or '' for child in root]
        need(title in labels and any(len(x) > 110 for x in labels),
             f'{stem}: accessible title/description absent')
        info = subprocess.run(['pdfinfo', str(pdf)], text=True, capture_output=True,
                              check=True).stdout
        need(re.search(r'^Pages:\s+1$', info, flags=re.MULTILINE) is not None and
             re.search(r'^Page size:\s+595\.\d+ x 841\.\d+ pts \(A4\)',
                       info, flags=re.MULTILINE) is not None,
             f'{stem}: PDF not one-page A4')
        fonts = subprocess.run(['pdffonts', str(pdf)], text=True, capture_output=True,
                               check=True).stdout
        need('DejaVuSans' in fonts and ' yes ' in fonts,
             f'{stem}: embedded DejaVu absent')
        words = subprocess.run(['pdftotext', str(pdf), '-'], text=True,
                               capture_output=True, check=True).stdout
        need(title in words and len(words.split()) > 55,
             f'{stem}: PDF selectable text absent')
        need(f'{stem}.svg' in alternatives and f'{stem}.pdf' in alternatives,
             f'{stem}: complete text alternative links absent')


def receipt(write: bool) -> None:
    paths = sorted((p for p in PACK.rglob('*') if p.is_file() and
                    p.name != 'MANIFEST.sha256' and
                    not {'__pycache__', '.ruff_cache'}.intersection(p.parts)),
                   key=lambda p: p.relative_to(PACK).as_posix())
    actual = {p.relative_to(PACK).as_posix() for p in paths}
    need(actual == REQUIRED,
         f'file inventory drift: missing={REQUIRED - actual}; extra={actual - REQUIRED}')
    expected = '\n'.join(f'{hashlib.sha256(p.read_bytes()).hexdigest()}  '
                         f'{p.relative_to(PACK).as_posix()}' for p in paths) + '\n'
    manifest = PACK / 'MANIFEST.sha256'
    if write:
        manifest.write_text(expected, encoding='utf-8')
    else:
        need(manifest.exists() and manifest.read_text(encoding='utf-8') == expected,
             'SHA-256 receipt absent or stale')


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument('--write-manifest', action='store_true')
    args = parser.parse_args()
    curriculum_and_rights()
    daily_structure()
    cases_checks_and_arithmetic()
    links = local_links()
    print_aids()
    receipt(args.write_manifest)
    print('PASS: QCAA Business 2025 v1.3 partial Unit 1 Topic 1; '
          'ten timed Days 1–10; thirty equivalent-target routes; '
          'two fresh public checks with arithmetic; '
          f'{links} valid local links/anchors; four A4 SVG/PDF/text aids; SHA-256 receipt')


if __name__ == '__main__':
    main()
