"""Class-6 numerical/text/boundary QA. No models, notebooks or network execution.

Run with Python 3 from any cwd. Writes only research/validation.json.
Baseline manifest is a snapshot taken before the class-6 edits, not a full-host audit.
"""
from pathlib import Path
from datetime import datetime, timezone
from decimal import Decimal
import hashlib
import json
import math
import re
import statistics
from urllib.parse import unquote

BASE = Path(__file__).resolve().parents[1]
COURSE = BASE.parent
WORKSPACE = COURSE.parents[1]
RESULT = BASE / 'research/validation.json'
checks = []
def check(name, actual, expected=None, tolerance=1e-10):
    if expected is None:
        ok = bool(actual)
    elif isinstance(actual, (int, float)) and isinstance(expected, (int, float)):
        ok = math.isclose(actual, expected, rel_tol=tolerance, abs_tol=tolerance)
    else:
        ok = actual == expected
    checks.append(dict(name=name, passed=ok, actual=actual, expected=expected))
    return actual

# Independent elementary derivations; these are not model measurements.
check('toolkit_seconds',64600/16000,4.0375)
check('crop_samples',round(2.56*32000),81920)
check('window_samples',round(.025*32000),800)
check('hop_samples',round(.010*32000),320)
check('conditional_complete_frames',1+(81920-800)//320,254)
check('grid_time',256//8,32)
check('grid_mel',128//32,4)
check('tokens',32*4,128)
check('patch_nominal_ms',8*10,80)
check('patch_first_last_ms',7*10,70)
check('patch_window_union_ms',25+7*10,95)
check('patch_params',768*8*32+768,197376)
asp=check('asp_params',(768*128+128)+(128+1),98561)
classifier=check('classifier_params',2*1536+(1536*256+256)+(256*2+2),397058)
check('backend_params',13+asp+classifier,495632)
check('initial_layer_weight',1/13,.07692307692307693)
check('native_jepa_hop_ms',10000/256,39.0625)
check('native_jepa_window_ms',2.5*10000/256,97.65625)
check('native_jepa_patch_ms',16*10000/256,625)
check('native_mae_patch_ms',16*10,160)
check('native_jepa_grid',(256//16,128//16),(16,8))
check('native_mae_tokens',(1024//16)*(128//16),512)
check('holdout_spoof',25380-600-20980,3800)
check('train_bona_fide',2580-600,1980)
check('train_spoof',22800-3800,19000)
w=check('weight_b',19000/1980,9.595959595959595)
check('weighted_equal_loss',(w*math.log(2)+math.log(2))/(w+1),math.log(2))
check('single_class_weight_cancels',(w*math.log(2))/w,math.log(2))
check('drop_last_batches',20980//32,655)
check('scheduled_iterations',6*(20980//32),3930)
check('example_exposures',6*(20980//32)*32,125760)
check('drop_false_iterations',6*math.ceil(20980/32),3936)
check('mask_union_percent',100*(1-(1-16/256)*(1-16/128)),17.96875)
tokens=[1,3]; weights=[.25,.75]
mu=sum(a*t for a,t in zip(weights,tokens))
var=sum(a*(t-mu)**2 for a,t in zip(weights,tokens))
check('asp_toy_mean',mu,2.5); check('asp_toy_variance',var,.75)
check('asp_toy_std',math.sqrt(var),.8660254037844386)
check('reverse_weights_mean',.75*1+.25*3,1.5)
check('logit_difference_A',7-6,1); check('logit_difference_B',4-0,4)
check('softmax_qB_A',1/(1+math.exp(-1)),.7310585786300049)
check('softmax_qB_B',1/(1+math.exp(-4)),.9820137900379085)
frozen=[8.74,8.59,8.07]; ft=[7.57,10.20,14.47]
check('dfadd_frozen_mean',statistics.mean(frozen),8.466666666666667)
check('dfadd_frozen_sample_sd',statistics.stdev(frozen),.351615320102399)
check('dfadd_ft_mean',statistics.mean(ft),10.746666666666668)
check('dfadd_ft_sample_sd',statistics.stdev(ft),3.4823315944)
check('dfadd_sd_ratio_round',round(statistics.stdev(ft)/statistics.stdev(frozen),2),9.9)
check('dfadd_mae_relative_reduction_pct',100*2.57/11.04,23.278985507246375)
check('dfadd_aasist_relative_reduction_pct',100*(41.87-8.47)/41.87,79.770718891808)
check('table3_gap_pp',float(Decimal('8.47')-Decimal('7.54')),.93)
check('table3_parameter_ratio',340/85.9,3.958090803259603)
check('table4_asv_ft_rounded_gap',float(Decimal('13.25')-Decimal('7.39')),5.86)
check('centered_token_rank_bound',min(64*128-1,768),768)
check('centered_utterance_rank_bound',min(64-1,768),63)
def erank(s):
    total=sum(s)
    return math.exp(-sum((v/total)*math.log(v/total) for v in s if v>0))
check('entropy_sigma_toy',erank([2,1]),1.88988157484231)
check('entropy_sigma_squared_toy',erank([4,1]),1.649384888466118)
check('anisotropy_label_toy_cos',9999/10001,.9998000199980002)
def adjacency(rows,cols):
    counts={k:0 for k in range(5)}
    for t in range(rows):
        for m in range(cols):
            k=(t>0)+(t<rows-1)+(m>0)+(m<cols-1)
            counts[k]+=1
    n=rows*cols; masked=n//2
    p=1-sum(counts[k]*math.comb(masked-1,k)/math.comb(n-1,k) for k in counts)/n
    return counts,p
counts,p=adjacency(32,4)
check('mask_grid_degree_counts',counts,{0:0,1:0,2:4,3:64,4:60})
check('mask_adjacency_toy',p,.9057952755905512)
check('mse_conditional_mean_toy',sum((t-0)**2 for t in [-1,1])/2,1)
check('mse_independent_sample_toy',sum((t-y)**2 for t in [-1,1] for y in [-1,1])/4,2)
combos=[(a,c) for a in [-1,1] for c in [-1,1]]
check('source_probe_toy_accuracy',sum(c==c for a,c in combos)/4,1)
check('artifact_head_toy_accuracy',sum((a>0)==(a>0) for a,c in combos)/4,1)
check('source_reliant_head_toy_accuracy',sum((a+2*c>0)==(a>0) for a,c in combos)/4,.5)

# Check the displayed main tables against the values directly verified from PDF renders.
lesson=(BASE/'05-DOC-BANG-KET-QUA.md').read_text(encoding='utf-8')
def row_cells(name):
    line=next(x for x in lesson.splitlines() if x.startswith('| '+name+' |'))
    return [x.strip() for x in line.strip('|').split('|')]
table2={
 'DFADD':([8.47,.35],[10.75,3.48],[41.87]),
 'ASV2019 LA eval':([17.70,1.44],[7.39,.69],[.83]),
 'ADD2022 T1':([], [44.10,2.04],[47.92]),
 'LibriSeVoc':([47.38,4.19],[41.96,2.93],[37.95]),
 'In-the-Wild':([47.55,8.55],[52.00],[43.02]),
}
for name,expected in table2.items():
    cells=row_cells(name)
    actual=tuple([float(x.replace(',','.')) for x in re.findall(r'\d+,\d+',c)] for c in cells[1:4])
    check('displayed_table2_'+name,actual,expected)
check('displayed_unrun_marker',row_cells('ADD2022 T1')[1],'—')
check('displayed_itw_n1','n=1' in row_cells('In-the-Wild')[2])
for name,gap in [('ASV fine-tuned',5.87),('ASV frozen',7.55),('DFADD frozen',2.57),('ADD fine-tuned',-2.08)]:
    cell=row_cells(name)[3].replace('−','-').replace(',','.')
    check('displayed_table4_reported_gap_'+name,float(cell),gap)

# Local link, text structure and encoding checks. Online validity is recorded by source review.
markdown=sorted(BASE.rglob('*.md'))+[COURSE/'00-BAT-DAU.md',COURSE/'06-GIAI-PHAU-PAPER.md']
broken=[]; structural=[]; link_count=0; external_count=0
pattern=re.compile(r'!?\[[^\]\n]*\]\((<[^>]+>|[^)\n]+)\)')
for path in markdown:
    s=path.read_text(encoding='utf-8')
    if '\ufffd' in s or any(0xe000<=ord(c)<=0xf8ff for c in s): structural.append(f'{path.name}: damaged Unicode')
    if '\\n\\nMean' in s: structural.append(f'{path.name}: literal newline')
    if s.count('<details>')!=s.count('</details>'): structural.append(f'{path.name}: details balance')
    if s.count('<summary>')!=s.count('</summary>'): structural.append(f'{path.name}: summary balance')
    if s.count('$$')%2: structural.append(f'{path.name}: display math balance')
    fence=None; table_width=None; previous=''
    for i,line in enumerate(s.splitlines(),1):
        match=re.match(r'^\s*(```|~~~)',line)
        if match:
            fence=None if fence==match[1] else match[1]
            previous=line; continue
        if fence: previous=line; continue
        if line.startswith('|'):
            width=len(re.split(r'(?<!\\)\|',line))
            if table_width is None: table_width=width
            elif width!=table_width: structural.append(f'{path.name}:{i}: table width')
        else: table_width=None
        if re.match(r'^\d+\.[^\s\d]',line): structural.append(f'{path.name}:{i}: list spacing')
        if re.match(r'^(?:\d+\. |[-*] )',line) and previous.strip() and not re.match(r'^(?:\d+\. |[-*] )',previous):
            structural.append(f'{path.name}:{i}: blank line before list')
        previous=line
    if fence: structural.append(f'{path.name}: unclosed fence')
    for match in pattern.finditer(s):
        target=match[1].strip('<> ')
        if re.match(r'^(https?://|app://|codex://|mailto:)',target): external_count+=1; continue
        if target.startswith('#'): continue
        target=unquote(target.split('#',1)[0])
        # Markdown optional quoted title is not used in the authored links.
        local=Path(target) if re.match(r'^[A-Za-z]:[/\\]',target) else path.parent/target
        link_count+=1
        if not local.exists(): broken.append(dict(source=str(path),target=target))
check('local_links_exist',not broken)
check('markdown_structure_utf8',not structural)
for n in range(1,11):
    p=next(BASE.glob(f'{n:02}-*.md'))
    s=p.read_text(encoding='utf-8')
    check(f'lesson_{n:02}_pedagogical_parts',all(x in s for x in ['Mục tiêu','tiền đề','<details>','Đào sâu']) and ('phản ví dụ' in s.lower()))

# Protect files already snapshotted at turn start. Do not read/write coordinator09 here.
baseline=json.loads((BASE/'research/baseline-hashes.json').read_text(encoding='utf-8'))
changed=[]; missing=[]
for name,sha in baseline.items():
    p=Path(name)
    if not p.exists(): missing.append(name)
    elif hashlib.sha256(p.read_bytes()).hexdigest()!=sha: changed.append(name)
check('protected_files_unchanged',not changed and not missing)
before=(WORKSPACE/'tmp/lop06/start-before.md').read_text(encoding='utf-8')
after=(COURSE/'00-BAT-DAU.md').read_text(encoding='utf-8')
paragraphs=[p for p in after.split('\n\n') if p.startswith('Lớp 6 có ')]
check('global00_exact_one_class6_paragraph',len(paragraphs)==1 and after.replace(paragraphs[0]+'\n\n','',1)==before)

report={
 'checked_at_utc':datetime.now(timezone.utc).isoformat(),
 'scope':'arithmetic/toys + displayed tables + local Markdown links/structure + selected protected hashes; no detector rerun',
 'passed':all(c['passed'] for c in checks),
 'check_count':len(checks),'checks':checks,
 'markdown_files_checked':len(markdown), 'local_links_checked':link_count,
 'external_links_listed_not_network_revalidated_by_script':external_count,
 'broken_links':broken,'structural_errors':structural,
 'protected_files_checked':len(baseline),'changed_protected_files':changed,'missing_protected_files':missing,
 'hash_exclusions':['coordinator09','allowed00/06','class6 outputs','tmp'],
 'pedagogical_status':'material QA only; pending learner answer M1',
}
RESULT.parent.mkdir(parents=True,exist_ok=True)
RESULT.write_text(json.dumps(report,ensure_ascii=False,indent=2),encoding='utf-8')
print(json.dumps({k:report[k] for k in ['passed','check_count','markdown_files_checked','local_links_checked','protected_files_checked','broken_links','structural_errors','changed_protected_files','missing_protected_files']},ensure_ascii=False,indent=2))
for c in checks:
    if not c['passed']: print('FAILED:',c['name'],repr(c['actual']),repr(c['expected']))
raise SystemExit(0 if report['passed'] else 1)
