#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Fail-closed local structural, source, maths, link, print and hash checks."""
import ast
import hashlib
import json
import re
import subprocess
import sys
from fractions import Fraction
from pathlib import Path

B=Path(__file__).resolve().parent
ROOT=B.parents[4]
errors=[]
def check(ok,msg):
    if not ok:errors.append(msg)
def read(name):return (B/name).read_text()

lessons=read('LESSONS.md');blocks=re.split(r'(?=^## Day \d+ ·)',lessons,flags=re.MULTILINE)[1:]
days=[];codes=set()
for block in blocks:
    head=block.split('\n',1)[0]
    m=re.match(r'## Day (\d+) ·',head)
    if not m:continue
    day=int(m.group(1));days.append(day);codes.update(re.findall(r'AC9[A-Z0-9]+',head))
    for word in ['**Question:**','**Kit:**','**Response move:**','**Another domain:**','**Home']:
        check(word in block,f'Day {day} missing {word}')
    check('**Routes:**' in block or '**Routes after collection:**' in block,f'Day {day} missing routes')
    times=['0–3','3–10','10–22','22–30','30–35'] if day not in (15,20) else ['0–3','3–8','8–20','20–30','30–35']
    for t in times:check(f'**{t}' in block,f'Day {day} missing timed stage {t}')
    route=re.search(r'\*\*Routes(?: after collection)?:\*\* (.*?)(?:\n|\*\*)',block)
    check(bool(route) and route.group(1).count(';')>=2,f'Day {day} fewer than 3 routes')
check(days==list(range(11,21)),f'Days should be 11–20, got {days}')
check('3 + 7 + 12 + 8 + 5' in lessons and '3 launch + 5 task briefing + 12 independent + 10 review + 5 close' in lessons,'35-minute timing examples missing')
learner=read('LEARNER.md')
check([int(x) for x in re.findall(r'^## Day (\d+) ·',learner,re.MULTILINE)]==list(range(11,21)),'clean learner daily page incomplete')
check('STUDENT-CHECKS.md' not in learner and 'ANSWER-AND-NEXT.md' not in learner,'learner page leaks held-out/staff link')
checks=read('STUDENT-CHECKS.md');key=read('teacher/ANSWER-AND-NEXT.md')
check(len(re.findall(r'^## Check [AB] ·',checks,re.MULTILINE))==2,'two fresh checks required')
for label in ['Check A · Day 15','Check B · Day 20']:
    check(label in checks and label in key,f'{label} missing from learner/staff split')
check('21 LABELS' not in learner and '24 pockets held' not in learner,'learner daily packet leaks fresh numbers')

# Exact workbook pin and exact official wording, row and level in crosswalk.
expected='db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3'
fw=json.loads((ROOT/'data/frameworks/acara-v9.json').read_text())
check(fw['source_sha256']==expected,'canonical workbook hash changed')
wb=ROOT/'research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx'
check(wb.exists() and hashlib.sha256(wb.read_bytes()).hexdigest()==expected,'official workbook missing or changed')
records={r['code']:r for r in fw['records'] if r['record_type']=='content_description' and r['code']}
cross=read('CURRICULUM-CROSSWALK.md');listed=set(re.findall(r'^\| (AC9[A-Z0-9]+) \|',cross,re.MULTILINE))
check(listed==codes,f'crosswalk and lesson code mismatch: {sorted(listed^codes)}')
for c in codes:
    r=records.get(c);check(bool(r),f'invalid code {c}')
    if not r:continue
    level=r['attributes']['level'];right='Years 5 and 6' if c.startswith(('AC9TDE6','AC9AVA6')) else 'Year 6'
    check(level==right,f'{c} wrong level {level}')
    wording=r['attributes']['content_description'].replace('\n',' ').replace('|','\\|')
    check(f'| {c} | {level} | {r["source_row"]} | {wording} |' in cross,f'{c} row/wording drift')
    check('not' in cross.lower() or 'conditional' in cross.lower(),'partial boundary absent')

# Distinct fictional arithmetic and same-whole comparisons.
source=read('SOURCE-PACKET.md')
check('12 + 6 + 6 = 24' in source and 12+6+6==24,'source tally drift')
check('14 + 7 + 7 = 28' in key and 14+7+7==28,'Check A tally/key drift')
check('16 + 8 + 8 = 32' in key and 16+8+8==32,'Check B tally/key drift')
check([Fraction(n,24) for n in (6,8,12,18)]==[Fraction(1,4),Fraction(1,3),Fraction(1,2),Fraction(3,4)],'24-whole line incorrect')
check([Fraction(n,32) for n in (8,16,24)]==[Fraction(1,4),Fraction(1,2),Fraction(3,4)],'32-whole check incorrect')
check(Fraction(1,4)<Fraction(1,3)<Fraction(1,2)<Fraction(3,4),'fraction order incorrect')
check(all(24==a*b for a,b in [(2,12),(3,8),(4,6)]) and 25==5*5,'array arithmetic incorrect')

# Local links including explicit heading anchors.
links=0
def slug(s):
    return re.sub(r'\s+','-',re.sub(r'[^\w\s-]','',s.lower())).strip('-')
for path in B.rglob('*.md'):
    txt=path.read_text()
    for raw in re.findall(r'\[[^\]]*\]\(([^)]+)\)',txt):
        if raw.startswith(('https://','http://','mailto:')):continue
        filename,_,anchor=raw.partition('#')
        target=(path.parent/filename) if filename else path
        check(target.exists(),f'{path.relative_to(B)} broken local link {raw}');links+=1
        if anchor and target.exists() and target.suffix=='.md':
            headings=[slug(x) for x in re.findall(r'^#{1,6}\s+(.+)$',target.read_text(),re.MULTILINE)]
            check(anchor in headings,f'{path.relative_to(B)} broken heading anchor {raw}')

# Text-alternative, dimensions, extractability, font and SVG access metadata.
stems=['trial-poster','source-audit','array-and-fraction','test-method','recommendation-canvas']
alt=read('print/TEXT-ALTERNATIVES.md')
for stem in stems:
    svg=B/'print'/f'{stem}.svg';pdf=B/'print'/f'{stem}.pdf'
    check(svg.exists() and pdf.exists(),f'missing SVG/PDF {stem}')
    if not (svg.exists() and pdf.exists()):continue
    s=svg.read_text();check('viewBox="0 0 794 1123"' in s and 'role="img"' in s and '<title' in s and '<desc' in s,f'{stem} missing A4/access info')
    check(f'{stem}.svg' in alt and f'{stem}.pdf' in alt,f'{stem} lacks text alternative links')
    info=subprocess.run(['pdfinfo',str(pdf)],capture_output=True,text=True,check=False)
    check(info.returncode==0 and 'Pages:           1' in info.stdout and '595.276 x 841.89 pts (A4)' in info.stdout,f'{stem} not one-page A4')
    extracted=subprocess.run(['pdftotext',str(pdf),'-'],capture_output=True,text=True,check=False)
    check(extracted.returncode==0 and len(extracted.stdout.strip())>140,f'{stem} text extraction failed')
    fonts=subprocess.run(['pdffonts',str(pdf)],capture_output=True,text=True,check=False).stdout
    check('DejaVuSans' in fonts and 'yes yes yes' in fonts,f'{stem} font not embedded')
for p in B.rglob('*.py'):
    try:ast.parse(p.read_text())
    except SyntaxError as e:errors.append(f'{p.relative_to(B)} Python syntax: {e}')
    check('SPDX-License-Identifier: Apache-2.0' in p.read_text(),f'{p.relative_to(B)} licence line missing')
check((B/'CODE-LICENSE.txt').exists() and (B/'print/FONT-RIGHTS.md').exists(),'code/font notice missing')

manifest=json.loads(read('manifest.json'))
files={p.relative_to(B).as_posix():p for p in B.rglob('*') if p.is_file() and p.name!='manifest.json' and '__pycache__' not in p.parts}
check(set(files)==set(manifest['files']),f'manifest file-set drift: {sorted(set(files)^set(manifest["files"]))}')
for name,p in files.items():
    entry=manifest['files'].get(name,{})
    check(entry.get('sha256')==hashlib.sha256(p.read_bytes()).hexdigest() and entry.get('bytes')==p.stat().st_size,f'{name} hash/size drift')
print(f"{'PASS' if not errors else 'FAIL'}: {len(days)} × 35-minute integrated sessions; 2 fresh checks; {len(codes)} exact official rows; {len(stems)} A4 SVG/PDF/text aids; {links} local links; {len(files)} hashed files")
for e in errors:print(' - '+e)
sys.exit(bool(errors))
