#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Fail-closed local audit for Year 8 Weeks 3–4 optional five-area pack."""
import ast
import hashlib
import json
import re
import subprocess
import sys
import tempfile
import wave
from pathlib import Path

B=Path(__file__).resolve().parent;ROOT=B.parents[4];errors=[]
def check(ok,msg):
 if not ok:errors.append(msg)
menus=['SCIENCE.md','HASS.md','HPE.md','TECHNOLOGIES.md','ARTS.md'];codes=set()
for name in menus:
 txt=(B/name).read_text();blocks=re.split(r'(?=^### Day \d+ ·)',txt,flags=re.MULTILINE)[1:]
 days=[int(re.search(r'^### Day (\d+)',x).group(1)) for x in blocks]
 check(days==list(range(11,21)),f'{name} days incomplete: {days}')
 for day,block in zip(days,blocks):
  codes.update(re.findall(r'AC9[A-Z0-9]+',block.split('\n',1)[0]))
  for stage in ['0–2:','2–4:','4–10:','10–12:']:
   check(stage in block,f'{name} Day {day} timed stage missing {stage}')
  for label in ['**Kit:**','**Routes:**','**Response move:**','**Home:**']:
   check(label in block,f'{name} Day {day} missing {label}')
  m=re.search(r'\*\*Routes:\*\* (.*?) \*\*Response move:',block,re.DOTALL)
  check(bool(m) and m.group(1).count(';')>=2,f'{name} Day {day} fewer than three switchable routes')
arts=(B/'ARTS.md').read_text()
for form in ['Dance','Drama','Media Arts','Music','Visual Arts']:
 check(len(re.findall(rf'^### Day \d+ · {re.escape(form)}:',arts,re.MULTILINE))==2,f'Arts {form} needs two days')
for area in ['Geography','History','Civics','Economics']:
 check(area in (B/'HASS.md').read_text(),f'HASS missing {area}')
learner=(B/'LEARNER-COPY.md').read_text();assess=(B/'ASSESSMENT.md').read_text();key=(B/'TEACHER-KEY.md').read_text()
check([int(x) for x in re.findall(r'^## Day (\d+)',learner,re.MULTILINE)]==list(range(11,21)),'learner day copy incomplete')
check('ASSESSMENT.md' not in learner and 'TEACHER-KEY.md' not in learner,'learner page links to held-out/key')
for sub in ['Science','HASS','HPE','Technologies','Arts']:
 for letter in 'AB':
  check(bool(re.search(rf'^### {sub} {letter} ·',assess,re.MULTILINE)),f'{sub} {letter} fresh learner check missing')
  check(bool(re.search(rf'^### {sub} {letter}$',key,re.MULTILINE)),f'{sub} {letter} private answer missing')
alt=(B/'ALTERNATIVE-CONTEXTS.md').read_text();rows=re.findall(r'^\| (\d{2}) \|(.+)\|$',alt,re.MULTILINE)
check([int(n) for n,_ in rows]==list(range(11,21)),'50 swap-in context rows incomplete')
for n,row in rows:check(len(row.split('|'))==5,f'Day {n} fewer than five subject swaps')

# Official curriculum hash, level, row and exact description.
hash0='db446882d2c00cf7c085a03e250e2442fda6c44011fc114680c46c1dc7a822c3'
fw=json.loads((ROOT/'data/frameworks/acara-v9.json').read_text());check(fw['source_sha256']==hash0,'canonical source hash drift')
wb=ROOT/'research/sources/acara-australian-curriculum-v9-download-2026-09-29.xlsx'
check(wb.exists() and hashlib.sha256(wb.read_bytes()).hexdigest()==hash0,'official workbook missing/hash drift')
rec={r['code']:r for r in fw['records'] if r['record_type']=='content_description' and r['code']}
cross=(B/'CURRICULUM-CROSSWALK.md').read_text();listed=set(re.findall(r'^\| (AC9[A-Z0-9]+) \|',cross,re.MULTILINE))
check(codes==listed,f'crosswalk/menu mismatch {sorted(codes^listed)}')
for c in codes:
 r=rec.get(c);check(bool(r),f'unknown content code {c}')
 if not r:continue
 level=r['attributes']['level'];expected='Year 8' if c.startswith(('AC9S8','AC9HG8','AC9HH8','AC9HC8','AC9HE8')) else 'Years 7 and 8'
 check(level==expected,f'{c}: incorrect level {level}')
 desc=r['attributes']['content_description'].replace('\n',' ').replace('|','\\|')
 check(f'| {c} | {level} | {r["source_row"]} | {desc} |' in cross,f'{c}: official row/wording drift')

# Local references and heading anchors.
links=0
def slug(s):return re.sub(r'\s+','-',re.sub(r'[^\w\s-]','',s.lower())).strip('-')
for p in B.rglob('*.md'):
 for raw in re.findall(r'\[[^\]]*\]\(([^)]+)\)',p.read_text()):
  if raw.startswith(('https://','http://','mailto:')):continue
  file,_,anchor=raw.partition('#');dest=(p.parent/file) if file else p
  check(dest.exists(),f'{p.relative_to(B)} broken link {raw}');links+=1
  if anchor and dest.exists() and dest.suffix=='.md':
   check(anchor in [slug(x) for x in re.findall(r'^#{1,6}\s+(.+)$',dest.read_text(),re.MULTILINE)],f'{p.relative_to(B)} broken anchor {raw}')

# Original reproducible A4 masters, accessible alternatives and font subset.
stems=['inquiry-mat','source-grid','access-movement-mat','design-data-mat','arts-process-page','binary-branch-grid']
textalt=(B/'print/TEXT-ALTERNATIVES.md').read_text()
for stem in stems:
 svg=B/'print'/f'{stem}.svg';pdf=B/'print'/f'{stem}.pdf'
 check(svg.exists() and pdf.exists(),f'{stem} SVG/PDF missing')
 if not svg.exists() or not pdf.exists():continue
 s=svg.read_text();check('viewBox="0 0 794 1123"' in s and 'role="img"' in s and '<title' in s and '<desc' in s,f'{stem} A4/access metadata missing')
 check(f'{stem}.svg' in textalt and f'{stem}.pdf' in textalt,f'{stem} text/tactile alternative missing')
 info=subprocess.run(['pdfinfo',str(pdf)],capture_output=True,text=True,check=False)
 check(info.returncode==0 and 'Pages:           1' in info.stdout and '595.276 x 841.89 pts (A4)' in info.stdout,f'{stem} PDF not single-page A4')
 out=subprocess.run(['pdftotext',str(pdf),'-'],capture_output=True,text=True,check=False)
 check(out.returncode==0 and len(out.stdout.strip())>110,f'{stem} PDF text not extractable')
 fonts=subprocess.run(['pdffonts',str(pdf)],capture_output=True,text=True,check=False).stdout
 check('DejaVuSans' in fonts and 'yes yes yes' in fonts,f'{stem} font not embedded')
check((B/'print/FONT-RIGHTS.md').exists() and (B/'CODE-LICENSE.txt').exists(),'separate font/code licence missing')

# Original audio and runnable offline bug lab.
with wave.open(str(B/'audio/layered-pulse.wav')) as w:
 check((w.getnchannels(),w.getsampwidth(),w.getframerate(),w.getnframes())==(1,2,16000,51200),'original WAV format/duration drift')
audiotxt=(B/'audio/TEXT-ALTERNATIVE.md').read_text()
check('Layer A, lower tap' in audiotxt and 'Layer B, higher tap' in audiotxt and '3.2 seconds' in audiotxt,'audio text/tactile equivalent missing')
code=B/'code/light_sample_flag.py';baseline=subprocess.run(['python3',str(code)],capture_output=True,text=True,check=False)
check(baseline.returncode==0 and baseline.stdout.count('match=False')==1 and baseline.stdout.count('match=True')==2,'starter deliberate bug or tests drift')
source=code.read_text();old='return "HIGH"  # BUG: expected PENDING_RECHECK for this case.'
check(source.count(old)==1,'single documented bug missing')
if old in source:
 with tempfile.TemporaryDirectory() as d:
  fixed=Path(d)/'fixed.py';fixed.write_text(source.replace(old,'return "PENDING_RECHECK"  # learner correction'))
  run=subprocess.run(['python3',str(fixed)],capture_output=True,text=True,check=False)
  check(run.returncode==0 and run.stdout.count('match=True')==3 and 'match=False' not in run.stdout,'proposed one-line fix did not close all tests')
for p in B.rglob('*.py'):
 try:ast.parse(p.read_text())
 except SyntaxError as e:errors.append(f'{p.relative_to(B)} syntax: {e}')
 check('SPDX-License-Identifier: Apache-2.0' in p.read_text(),f'{p.relative_to(B)} SPDX missing')

# Fictional arithmetic and distinction from real data.
check(1300-1000==300 and 1050-1000==50,'fictional town differences wrong')
check(25+18+12+5==60 and 20+14+6+10==50 and 20+14+6+8+2==50,'fictional budgets wrong')
check(14+7+7==28 and 16+8+8==32,'fresh-check totals wrong')
check(int('0111',2)==7 and int('1001',2)==9,'binary fresh-check key wrong')
check('not a real' in (B/'SOURCE-AND-RIGHTS.md').read_text().lower() or 'fictional' in (B/'SOURCE-AND-RIGHTS.md').read_text().lower(),'fictional source boundary missing')

manifest=json.loads((B/'manifest.json').read_text());files={p.relative_to(B).as_posix():p for p in B.rglob('*') if p.is_file() and p.name!='manifest.json' and '__pycache__' not in p.parts}
check(set(files)==set(manifest['files']),f'manifest file set drift {sorted(set(files)^set(manifest["files"]))}')
for name,p in files.items():
 e=manifest['files'].get(name,{})
 check(e.get('sha256')==hashlib.sha256(p.read_bytes()).hexdigest() and e.get('bytes')==p.stat().st_size,f'{name}: hash/size mismatch')
print(f"{'PASS' if not errors else 'FAIL'}: 5 menus × 10 days = 50 optional 12-minute blocks; 50 swap-ins; 10 fresh checks; {len(codes)} exact official rows; 6 A4 SVG/PDF/text aids; 1 original WAV; 1 offline code lab; {links} local links; {len(files)} hashed files")
for e in errors:print(' - '+e)
sys.exit(bool(errors))
