#!/usr/bin/env python3
"""Fail-closed local integrity and independent arithmetic audit for this starter."""
from __future__ import annotations

import argparse
import hashlib
import json
import re
import subprocess
import xml.etree.ElementTree as ET
from decimal import Decimal
from pathlib import Path
from urllib.parse import unquote

PACK = Path(__file__).resolve().parent
QCAA = 'https://www.qcaa.qld.edu.au/downloads/senior-qce/syllabuses/snr_accounting_25_syll.pdf'
STEMS = ('stakeholder-lens', 'ownership-cards', 'equation-mat',
         'statement-sorter', 'timing-bridge', 'comparison-grid')
REQUIRED = {
    'README.md', 'CURRICULUM-CROSSWALK.md', 'CASE-CARDS.md',
    'CURRENT-SOURCE-SLOT.md', 'LESSONS.md', 'LEARNER.md',
    'DAILY-CHOICES.md', 'DAILY-EXTRAS.md', 'EXAMPLE-BANK.md',
    'SPOKEN-PROMPTS.md', 'STUDENT-CHECKS.md',
    'SOURCE-AND-RIGHTS.md', 'QA-RUN-THROUGH.md',
    'teacher/ANSWER-AND-NEXT.md', 'verify_pack.py',
    'print/generate_print.py', 'print/TEXT-ALTERNATIVES.md',
    'print/FONT-RIGHTS.md', 'print/dejavu-font-copyright.txt',
}
for stem in STEMS:
    REQUIRED.update((f'print/{stem}.svg', f'print/{stem}.pdf'))


def need(condition: bool, message: str) -> None:
    if not condition:
        raise AssertionError(message)


def get(rel: str) -> str:
    return (PACK / rel).read_text(encoding='utf-8')


def scope() -> None:
    for rel in ('README.md', 'CURRICULUM-CROSSWALK.md',
                'SOURCE-AND-RIGHTS.md'):
        body = get(rel)
        need(QCAA in body and '2025 v1.4' in body,
             f'{rel}: QCAA version/link absent')
    cross = get('CURRICULUM-CROSSWALK.md').lower()
    for phrase in ('topic 1', 'topic 2', '55 hours', '250 minutes',
                   'not school instruments', 'queensland'):
        need(phrase in cross, f'curriculum boundary absent: {phrase}')
    readme = get('README.md').lower()
    for phrase in ('no classroom pilot', 'not a full 55-hour',
                   'six original a4', 'fictional', 'not qcaa school'):
        need(phrase in readme, f'readme boundary absent: {phrase}')
    rights = get('SOURCE-AND-RIGHTS.md')
    for identifier in ('| QCAA |', '| ASIC types |', '| AASB |',
                       '| A–F practice |', '| G fresh Day 5 |',
                       '| H/I fresh Day 10 |'):
        need(identifier in rights, f'ledger row absent: {identifier}')
    need('29 September 2026' in rights and 'CC BY 4.0' in rights,
         'source/rights date absent')


def daily() -> tuple[int, int]:
    lessons = get('LESSONS.md')
    choices = get('DAILY-CHOICES.md')
    extras = get('DAILY-EXTRAS.md')
    learner = get('LEARNER.md')
    spoken = get('SPOKEN-PROMPTS.md')
    for day in range(1, 11):
        found = re.findall(rf'^### Day {day} ·.*?(?=^### Day |\Z)', lessons,
                           flags=re.MULTILINE | re.DOTALL)
        need(len(found) == 1, f'Day {day}: missing/repeated lesson')
        section = found[0]
        need('**Target:**' in section and '**Prepare:**' in section,
             f'Day {day}: target or preparation absent')
        stages = (('Launch', 'Source access', 'Independent plan',
                   'Independent response', 'Self-audit', 'Submit')
                  if day in (5, 10) else
                  ('Launch', 'Model', 'Guided reading', 'Practice route',
                   'Audit', 'Exit'))
        durations = []
        for stage in stages:
            hits = re.findall(rf'\*\*{re.escape(stage)} · (\d+) min\.\*\*',
                              section)
            need(len(hits) == 1, f'Day {day}: {stage} missing/repeated')
            durations.append(int(hits[0]))
        need(durations == [2, 4, 5, 7, 4, 3],
             f'Day {day}: 25-minute timing drift {durations}')
        for rel, body in (('LEARNER.md', learner),
                          ('DAILY-CHOICES.md', choices),
                          ('DAILY-EXTRAS.md', extras),
                          ('SPOKEN-PROMPTS.md', spoken)):
            rows = re.findall(rf'^\| {day} \|(.+)$', body, flags=re.MULTILINE)
            need(len(rows) == 1, f'Day {day}: row absent/repeated in {rel}')
            cells = [x.strip() for x in rows[0].strip().strip('|').split('|')]
            count = (3 if rel == 'DAILY-CHOICES.md'
                     else 2 if rel == 'DAILY-EXTRAS.md' else 1)
            need(len(cells) == count and all(cells),
                 f'Day {day}: incomplete {rel}')
            if rel == 'DAILY-CHOICES.md':
                need(len(set(cells)) == 3, f'Day {day}: duplicate routes')
            if rel == 'DAILY-EXTRAS.md':
                need(all('→' in cell for cell in cells),
                     f'Day {day}: unworked swap')
    rows = re.findall(r'^\| [^|]+ \| [^|]+ \| [^|]+ \|$',
                      get('EXAMPLE-BANK.md'), flags=re.MULTILINE)
    need(len(rows) >= 20, f'only {len(rows)} everyday bridges')
    need('reading or handwriting support does not lower' in choices.lower(),
         'access/reasoning separation absent')
    return 30, 20


def checks() -> None:
    practice = get('CASE-CARDS.md')
    checks_text = get('STUDENT-CHECKS.md')
    key = get('teacher/ANSWER-AND-NEXT.md')
    for letter in 'ABCDEF':
        need(len(re.findall(rf'^## {letter} ·', practice,
                            flags=re.MULTILINE)) == 1,
             f'practice case {letter} absent/repeated')
    for letter in 'GHI':
        need(f'**New File {letter}.**' in checks_text,
             f'fresh File {letter} absent')
        need(f'## {letter} ·' not in practice,
             f'fresh File {letter} leaked into practice')
    for phrase in ('private phone', 'no bill due dates',
                   'no GST', 'unlisted public', 'provenance'):
        need(phrase.lower() in checks_text.lower(),
             f'check boundary missing: {phrase}')
    need('public by URL' in key and 'not secure QCAA' in key,
         'public/official assessment boundary absent')
    need(len(re.findall(r'^\| (?:[1-9]|10) \|', key,
                        flags=re.MULTILINE)) == 10,
         'daily feedback rows absent')
    for rel in ('LEARNER.md', 'DAILY-CHOICES.md', 'STUDENT-CHECKS.md'):
        need('ANSWER-AND-NEXT.md' not in get(rel),
             f'worked key exposed directly in learner flow: {rel}')


def fmt(value: Decimal) -> str:
    return f'${value:,.0f}'


def arithmetic() -> None:
    d = lambda n: Decimal(str(n))
    cards, checks_text, key, swaps = (get('CASE-CARDS.md'),
                                     get('STUDENT-CHECKS.md'),
                                     get('teacher/ANSWER-AND-NEXT.md'),
                                     get('DAILY-EXTRAS.md'))
    a_assets = d(9000) + d(5000) + d(1000)
    a_liabilities = d(4000) + d(1000)
    need(a_assets == a_liabilities + d(10000), 'A equation arithmetic')
    for value in (a_assets, a_liabilities, a_assets - a_liabilities):
        need(fmt(value) in cards, f'A derived amount absent: {value}')
    b_profit = d(1200) - (d(450) + d(150))
    b_operating = d(700) - d(450)
    b_cash = d(2000) + b_operating
    b_assets = b_cash + d(500)
    b_equity = d(2000) + b_profit
    need((b_profit, b_operating, b_cash, b_assets, b_equity,
          b_profit - b_operating) == tuple(map(d, (600, 250, 2250, 2750,
                                                   2600, 350))),
         'B independent arithmetic')
    need(b_assets == d(150) + b_equity, 'B closing equation')
    for value in (b_profit, b_operating, b_cash, b_assets, b_equity):
        need(fmt(value) in cards, f'B derived amount absent: {value}')
    c_profit = d(2400) - d(1100) - d(550)
    c_cash = d(1900) - d(1300) - d(500)
    need((c_profit, c_cash) == (d(750), d(100)), 'C arithmetic')
    for value in (c_profit, c_cash):
        need(fmt(value) in cards, f'C derived amount absent: {value}')
    for assets, liabilities, equity in (
        ((7000, 2000, 3000, 8000), (3000, 5000), 12000),
        ((10000, 5000, 7000, 18000), (6000, 14000), 20000),
    ):
        need(sum(map(d, assets)) == sum(map(d, liabilities)) + d(equity),
             'E position equation')
    g_assets = sum(map(d, (7600, 4200, 1100)))
    g_liabilities = d(3500) + d(900)
    need((g_assets, g_liabilities, g_assets - g_liabilities)
         == tuple(map(d, (12900, 4400, 8500))), 'G arithmetic')
    for value in (g_assets, g_liabilities, g_assets - g_liabilities):
        need(fmt(value) in key, f'G key derived amount absent: {value}')
    h_profit = d(1500) - d(700)
    h_operating = d(900) - d(500)
    h_cash = d(1600) + h_operating
    h_assets = h_cash + d(600)
    h_equity = d(1600) + h_profit
    need((h_profit, h_operating, h_cash, h_assets, h_equity)
         == tuple(map(d, (800, 400, 2000, 2600, 2400))),
         'H arithmetic')
    need(h_assets == d(200) + h_equity and
         h_profit - h_operating == d(600) - d(200),
         'H closing/timing equation')
    for value in (h_profit, h_operating, h_cash, h_assets, h_equity):
        need(fmt(value) in key, f'H key derived amount absent: {value}')
    i_assets = sum(map(d, (5000, 2000, 3000, 10000)))
    i_liabilities = d(2000) + d(6000)
    need((i_assets, i_liabilities, i_assets - i_liabilities)
         == tuple(map(d, (20000, 8000, 12000))), 'I arithmetic')
    for value in (i_assets, i_liabilities, i_assets - i_liabilities):
        need(fmt(value) in key, f'I key derived amount absent: {value}')
    need('**$12,900 = $4,400 + $8,500**' in key and
         '**$2,600 = $200 liabilities + $2,400 equity**' in key and
         '**$20,000 = $8,000 + $12,000**' in key,
         'key equations missing')
    for result in (750, 400, 280, 180, 470, 1100, 250, 900, 700):
        need(fmt(d(result)) in swaps,
             f'worked swap derived amount absent: {result}')
    need(fmt(d(12900)) not in checks_text,
         'Check A answer exposed in student file')


def slug(title: str) -> str:
    title = re.sub(r'<[^>]+>', '', title.lower())
    title = re.sub(r'[^\w\- ]', '', title)
    return title.replace(' ', '-')


def links() -> int:
    count = 0
    for md in PACK.rglob('*.md'):
        for url in re.findall(r'\[[^]]+\]\(([^)]+)\)',
                              md.read_text(encoding='utf-8')):
            if url.startswith(('https://', 'http://', 'mailto:')):
                continue
            base, _, anchor = unquote(url).partition('#')
            dest = (md.parent / base).resolve() if base else md
            need(dest.exists() and dest.is_relative_to(PACK),
                 f'broken/escaping link {md.relative_to(PACK)} -> {url}')
            if anchor and dest.suffix == '.md':
                heads = re.findall(r'^#{1,6} (.+)$',
                                   dest.read_text(encoding='utf-8'),
                                   flags=re.MULTILINE)
                need(anchor in {slug(h) for h in heads},
                     f'broken anchor {md.relative_to(PACK)} -> {url}')
            count += 1
    return count


def aids() -> None:
    alternatives = get('print/TEXT-ALTERNATIVES.md')
    normal = ' '.join(re.sub(r'[^\w]+', ' ', alternatives.lower()).split())
    need('tactile' in alternatives.lower() and
         'DejaVu' in get('print/FONT-RIGHTS.md'),
         'print access/rights absent')
    ns = {'s': 'http://www.w3.org/2000/svg'}
    for stem in STEMS:
        svg = PACK / f'print/{stem}.svg'
        pdf = PACK / f'print/{stem}.pdf'
        root = ET.parse(svg).getroot()
        need(root.attrib.get('width') == '210mm' and
             root.attrib.get('height') == '297mm', f'{stem}: not A4')
        need(root.find('s:title', ns) is not None and
             root.find('s:desc', ns) is not None,
             f'{stem}: SVG title/description absent')
        for element in root.findall('s:text', ns):
            words = ''.join(element.itertext()).strip()
            normalized = ' '.join(re.sub(r'[^\w]+', ' ', words.lower()).split())
            need(normalized in normal,
                 f'{stem}: visual text absent from exact alternative: {words}')
            need(int(element.attrib['x']) < 750 and
                 int(element.attrib['y']) < 1120,
                 f'{stem}: text origin outside page')
        extracted = subprocess.run(['pdftotext', str(pdf), '-'], check=True,
                                   capture_output=True, text=True).stdout
        need(len(extracted) > 300 and 'ACCOUNTING' in extracted,
             f'{stem}: PDF searchable text absent')
        info = subprocess.run(['pdfinfo', str(pdf)], check=True,
                              capture_output=True, text=True).stdout
        need('Pages:           1' in info and 'A4' in info,
             f'{stem}: PDF not one A4 page')
        need(stem.replace('-', ' ') in alternatives.lower(),
             f'{stem}: alternative absent')


def manifest(write: bool) -> int:
    paths = sorted(str(p.relative_to(PACK)) for p in PACK.rglob('*')
                   if p.is_file() and p.name != 'manifest.json'
                   and '__pycache__' not in p.parts)
    need(set(paths) == REQUIRED,
         f'file inventory drift: missing {sorted(REQUIRED-set(paths))}, '
         f'extra {sorted(set(paths)-REQUIRED)}')
    expected = {'schema_version': '1.0', 'files': {
        path: hashlib.sha256((PACK / path).read_bytes()).hexdigest()
        for path in paths}}
    target = PACK / 'manifest.json'
    if write:
        target.write_text(json.dumps(expected, indent=2, sort_keys=True) + '\n',
                          encoding='utf-8')
    else:
        need(target.exists(), 'SHA-256 manifest absent')
        need(json.loads(target.read_text(encoding='utf-8')) == expected,
             'SHA-256 manifest drift')
    return len(paths)


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument('--write-manifest', action='store_true')
    args = parser.parse_args()
    scope()
    routes, swaps = daily()
    checks()
    arithmetic()
    link_count = links()
    aids()
    files = manifest(args.write_manifest)
    print(f'PASS: 10 x 25 minutes; {routes} routes; {swaps} worked swaps; '
          f'2 fresh public checks; 6 A4 aid pairs; arithmetic; '
          f'{link_count} local links; {files} hashed source files')


if __name__ == '__main__':
    main()
