"""Check authored Markdown and toy arithmetic; never opens corpora or checkpoints."""
from pathlib import Path
import hashlib
import json
import math
import re
from datetime import datetime, timezone
from urllib.parse import unquote

HERE = Path(__file__).resolve().parents[1]
STUDY = HERE.parent
issues = []
numbers = {}

def check_number(name, actual, expected, tol=1e-9):
    numbers[name] = {"actual": actual, "expected": expected}
    if not math.isclose(actual, expected, rel_tol=tol, abs_tol=tol):
        issues.append(f"toy {name}: {actual} != {expected}")

def anchors(text):
    found = set(re.findall(r'<a\s+id="([^"]+)"', text))
    used = {}
    for h in re.findall(r'^#{1,6}\s+(.+)$', text, re.M):
        slug = re.sub(r'[^\w\- ]', '', h.lower()).replace(' ', '-')
        n = used.get(slug, 0)
        used[slug] = n + 1
        found.add(slug if not n else f"{slug}-{n}")
    return found

files = sorted(HERE.rglob('*.md')) + [STUDY / '00-BAT-DAU.md', STUDY / '02-DEEPFAKE-VA-BAI-TOAN.md']
link_count = 0
for path in files:
    txt = path.read_text(encoding='utf-8-sig')
    if '\ufffd' in txt:
        issues.append(f"replacement character: {path.name}")
    fence = None
    table_cols = None
    plain = []
    for line_no, line in enumerate(txt.splitlines(), 1):
        fm = re.match(r'^\s*(`{3,}|~{3,})', line)
        if fm:
            if fence is None:
                fence = fm.group(1)[0]
            elif fm.group(1)[0] == fence:
                fence = None
            table_cols = None
            continue
        if fence:
            continue
        plain.append(line)
        if line.startswith('|') and line.endswith('|'):
            cols = len(re.split(r'(?<!\\)\|', line)) - 2
            if table_cols is None:
                table_cols = cols
            elif table_cols != cols:
                issues.append(f"table columns: {path.name}:{line_no} {cols} vs {table_cols}")
        else:
            table_cols = None
    if fence:
        issues.append(f"unclosed fence: {path.name}")
    stripped = '\n'.join(plain)
    if len(re.findall(r'(?<!\\)\$', stripped)) % 2:
        issues.append(f"odd dollar delimiter count: {path.name}")
    for start, end in [(r'\[', r'\]'), (r'\(', r'\)')]:
        if stripped.count(start) != stripped.count(end):
            issues.append(f"unbalanced math delimiters {start}: {path.name}")
    for match in re.finditer(r'\[[^\]\n]+\]\((<[^>]+>|[^)\n]+)\)', stripped):
        target = match.group(1).strip('<>')
        if re.match(r'^(https?://|mailto:|codex:|app:)', target):
            continue
        link_count += 1
        link_path, _, fragment = unquote(target).partition('#')
        dest = Path(link_path) if re.match(r'^[A-Za-z]:[/\\]', link_path) else path.parent / link_path
        if not link_path:
            dest = path
        if not dest.exists():
            issues.append(f"missing link: {path.name} -> {target}")
        elif fragment and dest.suffix == '.md' and fragment not in anchors(dest.read_text(encoding='utf-8-sig')):
            issues.append(f"missing anchor: {path.name} -> {target}")

passport = (HERE / '12-HO-CHIEU-DATASET.md').read_text(encoding='utf-8')
required = ['Identity/version', 'Task/unit/labels', 'Genuine ancestry', 'Fake pipeline', 'Signal chain', 'Split/exposure', 'Generalization', 'Unknown/limits']
for section in re.split(r'^## P\d+ ', passport, flags=re.M)[1:]:
    for field in required:
        if f'| {field} |' not in section:
            issues.append(f"passport missing {field}: {section[:55]}")

# Recompute values from the stated toy inputs, independently of prose formatting.
check_number('AR_ABA_probability', .6 * .8 * .3, .144)
check_number('AR_ABA_NLL', -math.log(.6 * .8 * .3), 1.9379419794061366)
check_number('AR_BA_probability', .4 * .7, .28)
check_number('AR_BA_NLL', -math.log(.28), 1.2729656758128873)
check_number('GAN_D', (.9 - 1)**2 + .3**2, .10)
check_number('GAN_G', (.3 - 1)**2, .49)
check_number('GAN_D_changed', (.9 - 1)**2 + .8**2, .65)
check_number('GAN_G_changed', (.8 - 1)**2, .04)
for mu in (1, 2):
    check_number(f'VAE_KL_mu{mu}', .5 * (mu**2 + 1 - 1 - math.log(1)), mu**2 / 2)
normal_at_zero = 1 / math.sqrt(2 * math.pi)
check_number('flow_scale2_density', normal_at_zero / 2, .19947114020071635)
check_number('flow_scale3_density', normal_at_zero / 3, .1329807601338109)
corrupt = math.sqrt(.64) * 2 + math.sqrt(1 - .64) * -1
check_number('DDPM_corrupt', corrupt, 1)
check_number('DDPM_loss', (-1 - -.5)**2, .25)
check_number('DDPM_clean_estimate', (corrupt - math.sqrt(.36) * -.5) / math.sqrt(.64), 1.625)
corrupt2 = math.sqrt(.36) * -2 + math.sqrt(.64) * 1
check_number('DDPM_exercise_corrupt', corrupt2, -.4)
check_number('DDPM_exercise_estimate', (corrupt2 - math.sqrt(.64)) / math.sqrt(.36), -2)
check_number('FM_state', (1 - .25) * -1 + .25 * 3, 0)
check_number('FM_velocity', 3 - -1, 4)
check_number('FM_Euler', 0 + .1 * 3, .3)
check_number('FM_exercise_state', .25 * 2 + .75 * -2, -1)
check_number('FM_exercise_Euler', -1 + .2 * -3, -1.6)
for z, expected_sum in [(1.7, 1.5), (.6, .5)]:
    residual = z
    qs = []
    for book in ([0, 1, 2], [-.5, 0, .5]):
        q = min(book, key=lambda q: abs(residual - q))
        qs.append(q)
        residual -= q
    check_number(f'RVQ_sum_{z}', sum(qs), expected_sum)
    check_number(f'RVQ_residual_{z}', residual, z - expected_sum)
check_number('RVQ_error', (1.7 - 1.5)**2, .04)
check_number('codec_bitrate', 50 * 4 * math.log2(256), 1600)
check_number('codec_indices_2s', 2 * 50 * 4, 400)
check_number('codec_bitrate_K6', 50 * 6 * math.log2(256), 2400)
check_number('VALLE_nominal_bitrate', 75 * 8 * math.log2(1024), 6000)
for prior, marginal, posterior in [(.5, .5, .8), (.1, .26, 4/13), (.2, .32, .5)]:
    mix = .8 * prior + .2 * (1 - prior)
    check_number(f'label_shift_marginal_{prior}', mix, marginal)
    check_number(f'label_shift_posterior_{prior}', .8 * prior / mix, posterior)
check_number('covariate_shift_target_prior', .9 * .8 + .1 * .2, .74)
check_number('channel_toy_accuracy', (50 + 50) / (4 * 50), .5)

baseline = json.loads((HERE / 'research/baseline-hashes.json').read_text(encoding='utf-8-sig'))
allowed = {'00-BAT-DAU.md', '02-DEEPFAKE-VA-BAI-TOAN.md'}
changed, unchanged = [], []
for record in baseline:
    path = Path(record['path'])
    if not path.exists():
        issues.append(f"baseline file missing: {path}")
        continue
    now = hashlib.sha256(path.read_bytes()).hexdigest().upper()
    if now != record['sha256']:
        changed.append(str(path))
        if path.parent != STUDY or path.name not in allowed:
            issues.append(f"out-of-scope edit: {path}")
    else:
        unchanged.append(str(path))
report = {
    'timestamp_utc': datetime.now(timezone.utc).isoformat(),
    'status': 'PASS' if not issues else 'FAIL',
    'scope': 'Markdown structure, local navigation, toy arithmetic, original-file hash boundary; no empirical model/data validation',
    'internal_links_checked': link_count,
    'toy_calculations': numbers,
    'baseline_changed_allowed_files': changed,
    'baseline_unchanged_files': unchanged,
    'issues': issues,
    'limitations': ['No Markdown visual render or complete LaTeX parser', 'No corpus/checkpoint/exposure audit', 'Factual correctness relies on section-specific primary-source review recorded in source ledger']
}
(HERE / 'research/validation.json').write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding='utf-8')
print(json.dumps({'status': report['status'], 'issues': issues, 'changed_original_files': changed}, ensure_ascii=False, indent=2))
raise SystemExit(bool(issues))
