#!/usr/bin/env python3 """Build a credential-free, reproducible release from completed training artifacts.""" import ast import hashlib import json import shutil from collections import Counter, defaultdict from pathlib import Path HERE = Path(__file__).resolve().parent CODE = HERE.parent SCRATCH = Path('/e/scratch/reformo/schuhmann1_moss/whisper_score_regression') GEMINI = SCRATCH / 'gemini_finetune_preparation_20261005' OLD = SCRATCH / 'hf_layered_release/model' STUDY = SCRATCH / 'embedding_probe_study_20261006' OUT = SCRATCH / 'humaneness_ears_release_20261007/model' REPO = 'laion/humaneness-ears-base-medium' BENCHMARK = 'https://laion-whisper-base-small-emotion-voice-burst.static.hf.space/benchmark.html' def read(path): return json.loads(Path(path).read_text()) def write(path, value): path = Path(path) path.parent.mkdir(parents=True, exist_ok=True) path.write_text(json.dumps(value, indent=2, ensure_ascii=False, allow_nan=False) + '\n') def copy(source, relative): destination = OUT / relative destination.parent.mkdir(parents=True, exist_ok=True) if destination.suffix != '.pt' or not destination.exists() or destination.stat().st_size != Path(source).stat().st_size: shutil.copy2(source, destination) def dependency_closure(names): pending, seen = list(names), set() while pending: name = pending.pop() path = CODE / (name + '.py') if name in seen or not path.exists(): continue seen.add(name) for node in ast.walk(ast.parse(path.read_text())): if isinstance(node, ast.Import): modules = [n.name.split('.')[0] for n in node.names] elif isinstance(node, ast.ImportFrom): modules = [(node.module or '').split('.')[0]] else: continue pending.extend(m for m in modules if (CODE / (m + '.py')).exists()) return seen def stage(): OUT.mkdir(parents=True, exist_ok=True) for filename in ('LICENSE', 'classes.json', 'display_normalization.json'): copy(OLD / filename, filename) copy(GEMINI / 'prepared/checkpoint_normalization.json', 'training_normalization.json') copy(GEMINI / 'prepared/gemini_train_statistics.json', 'gemini_train_statistics.json') copy(GEMINI / 'prepared/gemini_burst_mapping.json', 'gemini_burst_mapping.json') copy(CODE / 'layered_multitask_model.py', 'model.py') shutil.copy2(HERE / 'inference.py', OUT / 'inference.py') copy(OLD / 'requirements.txt', 'requirements.txt') (OUT / 'requirements-training.txt').write_text('-r requirements.txt\npyarrow>=20\n') for size in ('base', 'small'): release_size = 'medium' if size == 'small' else size configuration = read(OLD / size / 'config.json') configuration.update(architectures=['LayeredMultiTaskWhisper'], is_encoder_decoder=False, use_cache=False, _name_or_path=REPO + '/' + release_size) write(OUT / release_size / 'config.json', configuration) copy(OLD / size / 'preprocessor_config.json', release_size + '/preprocessor_config.json') train = GEMINI / 'training' / ('whisper_' + size) copy(train / 'best.pt', release_size + '/checkpoint.pt') copy(train / 'config.json', 'training/configs/' + size + '_recorded_run.json') copy(train / 'metrics.jsonl', 'training/logs/' + size + '_epochs.jsonl') copy(train / 'COMPLETE.json', 'training/logs/' + size + '_COMPLETE.json') for kind in ['gemini_test_metrics', 'validation_metrics', 'public_metrics', 'legacy_on_flash_test_metrics', 'legacy_p3_test_metrics', 'legacy_p3_validation_metrics', 'gemini_p3_test_metrics', 'gemini_p3_validation_metrics', 'crema_matched_adapter', 'ravdess_matched_adapter']: copy(STUDY / 'whisper/gemini' / size / (kind + '.json'), 'evaluation/' + size + '/' + kind + '.json') copy(OLD / 'training' / (size + '_run_config.json'), 'training/original_curriculum/' + size + '_run_config.json') names = dependency_closure(['train_layered_curriculum', 'full_ladder_dataset', 'layered_multitask_model']) for name in sorted(names): copy(CODE / (name + '.py'), 'training/' + name + '.py') for name in ['common.py', 'data.py', 'train_whisper.py', 'train_whisper.sbatch', 'prepare.py', 'backfill.py', 'configure.py']: copy(CODE / 'gemini_finetune' / name, 'training/gemini_finetune/' + name) for name in ['annotate.py', 'gemini_voice_annotation_master_prompt_detailed.txt']: copy(CODE / 'gemini_s10_smoke' / name, 'training/gemini_s10_smoke/' + name) teachers = CODE.parent / 'vocal_burst_pool/p3' for name in ['annotate_dnsmos.py', 'annotate_audiobox.py', 'annotate_voiceclap.py', 'annotate_empathic.py']: copy(teachers / name, 'vocal_burst_pool/p3/' + name) for name in dependency_closure(['evaluate_public_benchmarks']): copy(CODE / (name + '.py'), 'training/' + name + '.py') for path in sorted((CODE / 'embedding_probe_study').glob('*.py')): copy(path, 'training/embedding_probe_study/' + path.name) helper = Path('/e/scratch/reformo/schuhmann1_moss/out/m2_600m_ladder/code_v4_highlr_20260920/tar_member_map.py') if helper.exists(): copy(helper, 'training/tar_member_map.py') copy(SCRATCH.parent / 'code/fastgen.sh', 'training/original_curriculum/fastgen.sh') copy(HERE / 'make_finetune_config.py', 'training/make_finetune_config.py') for name in ['export_and_verify.py', 'export_rank.py', 'verify.sbatch', 'package_release.py', 'publish_release.py']: copy(HERE / name, 'release_tools/' + name) copy(SCRATCH / 'gemini_multisource_100h_20261004/combined_summary.json', 'training/provenance/selected_pool.json') copy(GEMINI / 'prepared/TRAINING_READY.json', 'training/provenance/TRAINING_READY.json') (OUT / 'NOTICE.md').write_text('''# License and attribution Author: **Christoph Schuhmann**. Organization: **LAION**. Release: 7 October 2026. LAION's new fine-tuned weights, code and documentation: **CC BY 4.0**, see `LICENSE`. Please credit Christoph Schuhmann and LAION, link this repository and indicate changes. OpenAI Whisper's original code and weights retain their MIT notice: https://github.com/openai/whisper/blob/main/LICENSE Hugging Face Transformers is an external Apache-2.0 dependency: https://github.com/huggingface/transformers/blob/main/LICENSE Teacher models and source audio retain their own terms; they are not redistributed here. Orange's identity teacher card declares CC BY-SA 3.0: https://huggingface.co/Orange/Speaker-wavLM-id Orange timbre: https://huggingface.co/Orange/Speaker-wavLM-tbr The release license is not a grant of rights over upstream datasets or packages. ''') (OUT / '.gitattributes').write_text('*.safetensors filter=lfs diff=lfs merge=lfs -text\n*.pt filter=lfs diff=lfs merge=lfs -text\n') print('STAGED', str(OUT), 'training dependency modules', len(names), flush=True) def data_summary(): summary = OUT / 'training/provenance/valid_split_summary.json' if summary.exists(): return read(summary) counts, hours, languages, sources = Counter(), defaultdict(float), Counter(), Counter() fingerprint = hashlib.sha256() with (GEMINI / 'prepared/targets.jsonl').open('rb') as stream: for raw in stream: fingerprint.update(raw) row = json.loads(raw) if not row['ready_for_whisper_training']: continue counts[row['split']] += 1 hours[row['split']] += row['duration_s'] / 3600 languages[row['annotation']['language']] += 1 for source in {m['source'] for m in row['source_memberships']}: sources[source] += 1 assert counts == {'train': 58380, 'validation': 3239, 'test': 3544}, counts result = {'valid_clips': sum(counts.values()), 'split_clips': dict(counts), 'split_hours': dict(hours), 'valid_total_hours': sum(hours.values()), 'language_strings': dict(languages), 'source_membership_counts_nonexclusive': dict(sources), 'prepared_targets_sha256': fingerprint.hexdigest(), 'deduplication': 'Exact selected compressed audio SHA256 across all selected sources/tasks', 'split': '90/5/5 deterministic group hash; explicit speaker or synthetic family before exact audio SHA', 'excluded_clips': 1036, 'captions_used_as_model_inputs': False} write(summary, result) return result def card(): data = data_summary() template = (HERE / 'README_template.md').read_text() base = read(OUT / 'evaluation/base/gemini_test_metrics.json') small = read(OUT / 'evaluation/small/gemini_test_metrics.json') def num(v): return 'Unavailable' if v is None else f'{v:.4f}' def score(m, name, key): return next(x[key] for x in m['per_score'] if x['score'] == name) def family(m, prefix): values = [x['normalized_mae'] for x in m['per_score'] if x['score'].startswith(prefix) and x['normalized_mae'] is not None] return sum(values) / len(values) rows = [] for task, function in [ ('Burst frame F1 ↑', lambda m: m['frame_f1']), ('Burst event F1, IoU ≥0.5 ↑', lambda m: m['burst_localization']['0.5']['f1']), ('Burst start MAE on IoU ≥0.1 matches, seconds ↓', lambda m: m['burst_start_mae_s']), ('Burst end MAE on IoU ≥0.1 matches, seconds ↓', lambda m: m['burst_end_mae_s']), ('53-class accuracy with reference spans ↑', lambda m: m['burst_class_oracle_span_accuracy']), ('53-class accuracy on matched predicted spans ↑', lambda m: m['burst_matched_event_class_accuracy']), ('Orange timbre cosine ↑ (1,718 valid single-speaker clips)', lambda m: m['speaker_cosine']['timbre']['mean']), ('Orange identity cosine ↑ (1,718 valid single-speaker clips)', lambda m: m['speaker_cosine']['identity']['mean']), ('CPS raw MAE, characters/second ↓', lambda m: m['cps_raw_mae']), ('Genuineness raw MAE on 0–6 scale ↓ (3,544 clips)', lambda m: score(m, 'genuineness_0_6', 'raw_mae')), ('Genuineness normalized MAE ↓', lambda m: score(m, 'genuineness_0_6', 'normalized_mae')), ('Blend raw MAE on 0–10 scale ↓ (2,391 valid clips)', lambda m: score(m, 'blend_0_10', 'raw_mae')), ('Blend normalized MAE ↓', lambda m: score(m, 'blend_0_10', 'normalized_mae')), ('40 emotions, mean normalized MAE ↓', lambda m: family(m, 'emo_')), ('57 VoiceNet dimensions, mean normalized MAE ↓', lambda m: family(m, 'vn_')), ('AudioBox four axes, mean normalized MAE ↓', lambda m: family(m, 'audiobox:')), ('DNSMOS seven outputs, mean normalized MAE ↓', lambda m: family(m, 'dnsmos:'))]: rows.append('| ' + task + ' | ' + num(function(base)) + ' | ' + num(function(small)) + ' |') holdout_table = '| Flash test target / metric | Base | Medium |\n| --- | ---: | ---: |\n' + '\n'.join(rows) public_rows = [] for task, key, metric in [('EmoNet-Voice intensity, Pearson r ↑', 'emonet', 'pearson'), ('EmoNet-Voice intensity, Spearman ρ ↑', 'emonet', 'spearman'), ('VoiceNet-Emo, mean prompt Spearman ρ ↑', 'emolia-emo', 'mean_prompt_spearman'), ('VoiceNet-Ext / emolia-dim, mean prompt Spearman ρ ↑', 'emolia-dim', 'mean_prompt_spearman')]: values = [] for size in ['base', 'small']: result = read(OUT / 'evaluation' / size / 'public_metrics.json')[key] values.append(result['all_mapped_40' if key == 'emonet' else 'repo_min_2_raters'][metric]) public_rows.append('| ' + task + ' | ' + num(values[0]) + ' | ' + num(values[1]) + ' |') for key in ['crema', 'ravdess']: values = [read(OUT / 'evaluation' / s / (key + '_matched_adapter.json'))['metrics']['accuracy'] for s in ['base', 'small']] public_rows.append('| ' + ('CREMA-D' if key == 'crema' else 'RAVDESS') + ' matched actor-CV score-MLP accuracy ↑ | ' + num(values[0]) + ' | ' + num(values[1]) + ' |') public_table = '| Human-label benchmark / metric | Base | Medium |\n| --- | ---: | ---: |\n' + '\n'.join(public_rows) target_rows = [] for index, (b, s) in enumerate(zip(base['per_score'], small['per_score'])): assert b['score'] == s['score'] target_rows.append('| ' + str(index) + ' | `' + b['score'] + '` | ' + str(b['n']) + ' | ' + ' | '.join( num(m[k]) for m in [b, s] for k in ['raw_mae', 'normalized_mae', 'pearson', 'spearman']) + ' |') target_table = '| Index | Exact target key | Valid N | Base raw MAE | Base norm. MAE | Base r | Base ρ | Medium raw MAE | Medium norm. MAE | Medium r | Medium ρ |\n| ---: | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |\n' + '\n'.join(target_rows) names = read(OUT / 'classes.json')['names'] classes = '| ID | Canonical vocal-burst class |\n| ---: | --- |\n' + '\n'.join('| ' + str(i) + ' | ' + name + ' |' for i, name in enumerate(names)) card = template.replace('{{HOLDOUT_TABLE}}', holdout_table).replace('{{PUBLIC_TABLE}}', public_table).replace('{{TARGET_TABLE}}', target_table).replace('{{CLASS_TABLE}}', classes).replace('{{VALID_HOURS}}', f"{data['valid_total_hours']:.2f}") assert '{{' not in card.replace('author = {Schuhmann, Christoph and {LAION}}', ''), 'Unexpanded placeholder' (OUT / 'README.md').write_text(card) print('WROTE model card', len(card.encode()), flush=True) if __name__ == '__main__': stage() card()