File size: 8,166 Bytes
4be6a52
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
# Independent final evidence audit

This archives the exact CPU audit script run after evaluation. Its SHA256 is
recorded in [the audit result](final-audit.json). It assumes the original local
`runs/`, dataset and checkpoint layout; it is an execution record, not a portable
release-verification CLI. Use `scripts/verify_release.py` for release downloads.
No new games or model inference were run by this audit.

```text
"""Independent CPU-only audit of frozen final evidence; reconstructs saved actions only."""
import hashlib
import importlib.util
import json
import math
import random
import statistics
import time
from pathlib import Path
from types import SimpleNamespace

ROOT = Path.cwd()
OUT = ROOT / 'runs/final-review/audit.json'
assert not OUT.exists(), 'Refuse to overwrite audit'
start = time.monotonic()
def read(p):
    return json.loads(Path(p).read_text())
def sha(p):
    h = hashlib.sha256()
    with Path(p).open('rb') as f:
        while chunk := f.read(1048576):
            h.update(chunk)
    return h.hexdigest()
def module(name):
    spec = importlib.util.spec_from_file_location(name, ROOT / 'scripts' / name)
    obj = importlib.util.module_from_spec(spec)
    spec.loader.exec_module(obj)
    return obj

evaluator = module('evaluate_clef.py')
exporter = module('export_demo.py')
request = read('runs/final-evaluation/request.json')
report_path = Path('runs/final-evaluation/report.json')
report = read(report_path)
print('Loaded final report; validating frozen identities.', flush=True)
players = ['base', 'base-fp32', 'trained', 'random', 'heuristic']
seeds = list(range(30000, 30200))
assert request['players'] == players
assert set(report['players']) == set(players)
for key, expected in [('mode', 'tournament'), ('final_test', True), ('max_pieces', 200), ('seeds', seeds)]:
    assert report[key] == request[key] == expected, key
for file, expected in request['source_hashes'].items():
    assert sha(file) == expected, file
assert report['provenance']['source_hashes'] == request['source_hashes']
assert report['provenance']['installed_versions'] == request['installed_versions']
checkpoint = Path(request['checkpoint'])
checkpoint_hashes = evaluator.checkpoint_hashes(checkpoint)
assert checkpoint_hashes == request['checkpoint_sha256'] == report['checkpoint_sha256']
selection_path = Path(request['selection_file'])
assert sha(selection_path) == request['selection_sha256']
assert read(selection_path) == report['selection']
args = SimpleNamespace(seeds=tuple(seeds), final_test=True, max_pieces=200, players=players,
                       selection_file=selection_path, checkpoint=checkpoint, max_length=4096)
evaluator.validate_selection(args, checkpoint_hashes)
evaluator.validate_neural_runtimes(report['players'])
episodes = report['episodes']
assert len(episodes) == 1000
indexed = {(e['player_id'], e['seed']): e for e in episodes}
assert len(indexed) == 1000
assert set(indexed) == {(p, s) for p in players for s in seeds}
metrics = ('lines', 'score', 'pieces')
summaries = {}
episode_hashes = {}
def summary(values):
    if not values:
        return dict(count=0, mean=None, median=None, p95=None)
    return dict(count=len(values), mean=statistics.mean(values), median=statistics.median(values),
                p95=sorted(values)[math.ceil(.95 * len(values)) - 1])
for player in players:
    metadata = read(Path('runs/final-evaluation') / player / 'player.json')
    assert metadata == report['players'][player]
    rows = []
    for seed in seeds:
        path = Path('runs/final-evaluation') / player / f'seed-{seed}.json'
        episode = read(path)
        assert episode == indexed[player, seed], str(path)
        assert episode['player']['revision'] == metadata['revision']
        assert episode['player']['runtime_config'] == metadata['runtime_config']
        assert episode['max_pieces'] == 200
        exporter.validated_player(episode, 0)
        if player in ('base', 'base-fp32', 'trained'):
            assert all(d['probabilities'] and 0 < d['input_tokens'] <= 4096 for d in episode['decisions'])
        episode_hashes[str(path)] = sha(path)
        rows.append(episode)
    primary = {m: summary([0 if e['errors'] else e['outcome'][m] for e in rows]) for m in metrics}
    observed = {m: summary([e['outcome'][m] for e in rows]) for m in metrics}
    failed = [e['seed'] for e in rows if e['errors']]
    cap_count = sum(e['outcome']['cap_hit'] for e in rows)
    events = [d for e in rows for d in e['decisions']]
    computed = dict(episodes=200, failed_episodes=len(failed), error_rate=len(failed)/200,
                    invalid_decisions=sum(er['kind']=='invalid_decision' for e in rows for er in e['errors']),
                    cap_hit_rate=cap_count/200, failure_adjusted=primary, observed_before_error=observed,
                    latency_seconds=summary([d['decision_seconds'] for d in events]),
                    input_tokens=summary([d['input_tokens'] for d in events if d.get('input_tokens') is not None]),
                    failed_seeds=failed)
    for comparison in ('trained_vs_base', 'trained_vs_heuristic', 'trained_vs_base_fp32', 'base_fp32_vs_base'):
        stored = report[comparison]['players'][player]
        for key, value in computed.items():
            assert stored[key] == value, (comparison, player, key)
    summaries[player] = {**computed, 'cap_count': cap_count}
    print(f'{player}: all 200 saved replays/observations/decisions verified; mean lines={primary["lines"]["mean"]}', flush=True)
comparisons = {}
for key, left, right in [('trained_vs_base','trained','base'), ('trained_vs_heuristic','trained','heuristic'),
                          ('trained_vs_base_fp32','trained','base-fp32'), ('base_fp32_vs_base','base-fp32','base')]:
    stored = report[key]
    assert stored['trained_id'] == left and stored['base_id'] == right
    assert stored['seeds'] == seeds and stored['matches_reserved_test_pool'] is True
    assert stored['max_pieces'] == 200 and stored['selection_performed'] is False
    assert stored['max_error_rate'] == 0
    comparisons[key] = {}
    for metric in metrics:
        def score(p, seed):
            e = indexed[p, seed]
            return 0 if e['errors'] else e['outcome'][metric]
        deltas = [score(left,s)-score(right,s) for s in seeds]
        rng = random.Random(2026)
        means = sorted(sum(rng.choice(deltas) for _ in seeds)/200 for _ in range(10000))
        def percentile(q):
            n = 9999*q
            low = int(n)
            return means[low] + (means[math.ceil(n)]-means[low])*(n-low)
        expected = dict(episodes=200, mean_difference=statistics.mean(deltas), ci95_lower=percentile(.025),
                        ci95_upper=percentile(.975), method='paired episode percentile bootstrap',
                        bootstrap_samples=10000, bootstrap_seed=2026)
        assert expected == stored['paired_trained_minus_base'][metric], (key, metric)
        comparisons[key][metric] = expected
    acceptable = not summaries[left]['failed_seeds'] and not summaries[right]['failed_seeds']
    assert stored['errors_acceptable'] == acceptable
    assert stored['positive_mean_lines_signal'] == (comparisons[key]['lines']['ci95_lower'] > 0 and acceptable)
    print(f'{key}: all three paired metrics and bootstrap intervals independently match.', flush=True)
result = dict(status='passed', audit_script_sha256=sha(__file__), elapsed_seconds=time.monotonic()-start,
              report_sha256=sha(report_path), request_sha256=sha('runs/final-evaluation/request.json'),
              selection_sha256=sha(selection_path), checkpoint_sha256=checkpoint_hashes,
              source_hashes=request['source_hashes'], episode_file_sha256=episode_hashes,
              episodes_verified=1000, summaries=summaries, comparisons=comparisons,
              method='Saved replay reconstruction; independent statistics and random.choice paired bootstrap. No new policy decisions/model loads.')
OUT.write_text(json.dumps(result, indent=2, allow_nan=False)+'\n')
print(json.dumps({k: result[k] for k in ('status','elapsed_seconds','report_sha256','selection_sha256','episodes_verified')}), flush=True)

```