Audio8-ASR-Infinite-Compressed / sources /scripts /probe_published_q3_gpu_long_v1.py
Reza2kn's picture
Reduce MLX cache-rebase allocation and report sustained GPU streaming
fddfb10 verified
Raw History Blame Contribute Delete
6.18 kB
"""Long exposed-case check of the shipped CUDA/Vulkan launchers, no model selection."""
from pathlib import Path
import argparse
import json
import os
import sys
import time
BASE = Path('/home/rezo/audio8-q4-recovery-20260924')
CONSUMER = Path('/dev/shm/audio8-q3-public-linux-consumer-v1')
sys.path.insert(0, '/dev/shm/audio8-q3-exact-gpu-v1')
import probe_q3_exact_gguf_smoke_v1 as helper
def main():
p = argparse.ArgumentParser(description=__doc__)
p.add_argument('--backend', choices=['cuda', 'vulkan'], required=True)
p.add_argument('--device', type=int, choices=[0], default=0)
p.add_argument('--output', type=Path, required=True)
args = p.parse_args()
assert not args.output.exists() and 'LD_LIBRARY_PATH' not in os.environ
for name, digest in helper.PINNED.items():
assert helper.sha(helper.ROOT / name) == digest, name
manifest_path = BASE / 'native-caption-observer-v1/data/continuous-v1/manifest.json'
manifest = json.loads(manifest_path.read_text())
case = next(c for c in manifest['cases'] if c['id'] == 'ami-IS1000a')
case['path'] = str(BASE / 'native-heldout-campaign-v1/inputs/IS1000a.f32le')
assert case['sha256'] == 'a2ca3b651137e316ec19127b5f4c785c258673968a93cd97df0834f97fcd125f'
assert case['samples'] == 5115440 and helper.sha(case['path']) == case['sha256']
release_path = CONSUMER / 'release_manifest.json'
release = json.loads(release_path.read_text())
records = {r['path']: r for r in release['files']}
names = [f'run-{args.backend}.sh', f'gguf/bin/linux-x86_64-{args.backend}/audio8-gguf',
f'gguf/bin/linux-x86_64-{args.backend}/libaudio8-bridge.so',
'gguf/audio8-q3-exact-q4_0.gguf', 'assets/tokenizer.a8tok', 'assets/silero_vad.onnx']
pins = {str(CONSUMER / name): records[name]['sha256'] for name in names}
assert pins[str(CONSUMER / 'gguf/audio8-q3-exact-q4_0.gguf')] == helper.MODEL_SHA
segments_path = BASE / 'original-cuda-reference-v1/reports/silero-is1000a-causal-segments-v1.json'
expected_segments = json.loads(segments_path.read_text())['segments']
dependencies = [Path(__file__), Path(helper.__file__), release_path, manifest_path,
segments_path, Path(case['path']), *[helper.ROOT / n for n in helper.PINNED]]
pins.update({str(path): helper.sha(path) for path in dependencies})
assert all(helper.sha(path) == digest for path, digest in pins.items())
args.output.mkdir(parents=True)
command = ['/bin/sh', str(CONSUMER / f'run-{args.backend}.sh'), '-', 'en', str(args.device)]
plan = dict(status='frozen_before_inference', pins=pins, command=command, case=case,
purpose='Already evaluated full meeting; backend regression, not new heldout qualification',
fresh_heldout=False, production_ready=False, peak_device_memory_measured=False)
(args.output / 'plan.json').write_text(json.dumps(plan, indent=2) + '\n')
paths = helper.observer.output_paths(args.output / 'run.json')
report = dict(status='running', started_unix=time.time(), plan_sha256=helper.sha(args.output / 'plan.json'))
paths['report'].write_text(json.dumps(report, indent=2) + '\n')
try:
rows, observations, feeder = helper.observer.capture_process(command, Path(case['path']),
case['samples'], paths, timeout_seconds=1200, max_events=20000,
ready_timeout=120, ready_kind='caption_stream_ready')
table = helper.token_table(CONSUMER / 'assets/tokenizer.a8tok')
report.update(helper.validate(rows, case, table, args.backend, args.device, 1))
starts = [r for r in rows if r['kind'] == 'caption_segment_start']
ends = [r for r in rows if r['kind'] == 'caption_segment_end']
actual = [dict(id=s['segment'], start_sample=s['start_sample'], decision_sample=s['decision_sample'],
end_sample=e['end_sample'], final_eof=e['final_eof']) for s, e in zip(starts, ends, strict=True)]
assert len(actual) == len(expected_segments)
for a, e in zip(actual, expected_segments, strict=True):
assert all(a[k] == e[k] for k in ('start_sample', 'decision_sample', 'end_sample', 'final_eof'))
deltas = []
for row, observation in zip(rows, observations, strict=True):
assert row['kind'] == observation['kind'] and row.get('index') == observation['index']
if row['kind'] == 'caption_text_delta':
deltas.append(dict(text=row['text'], index=row['index'],
seconds=observation['received_monotonic_seconds'] - feeder['feeder_started_monotonic']))
mature = [r for r in rows if r['kind'] == 'caption_window' and not r['eof_drain']
and r['segment_index'] >= 375]
assert mature
report.update(status='complete', feeder=feeder, consumer_deltas=deltas,
caption_score=helper.caption_score(case, deltas, actual_arrival_paced=True,
timing_basis='Host receipt relative to post-ready20ms source feeder; complete exposed meeting'),
mature_windows=len(mature), mature_work_rtf=sum(r['work_seconds'] for r in mature)/(len(mature)*.08),
memory=helper.observer.parse_time(paths['timing'].read_text(), 'Linux'),
peak_device_memory_measured=False, frozen_segments_exact=True,
invoked_shipped_launcher=True, release_qualification=False, production_ready=False)
assert all(helper.sha(path) == digest for path, digest in pins.items())
report['all_assets_unchanged'] = True
except BaseException as error:
report.update(status='failed', error=f'{type(error).__name__}: {error}')
raise
finally:
report['finished_unix'] = time.time()
report['logs'] = {name: helper.asset(path) for name, path in paths.items()
if name != 'report' and path.exists()}
paths['report'].write_text(json.dumps(report, indent=2) + '\n')
print(json.dumps(dict(status=report['status'], mature_work_rtf=report['mature_work_rtf'],
counts=report['caption_score']['counts'])))
if __name__ == '__main__':
main()