Download release_tools/package_release.py from laion/humaneness-ears-base-medium: direct link, hf CLI and curl.
- Browser
- Download file 13.7 kB
-
https://huggingface.co/laion/humaneness-ears-base-medium/resolve/main/release_tools/package_release.py
- Command line
-
hf download hf://laion/humaneness-ears-base-medium/release_tools/package_release.py
-
curl -L -o package_release.py https://huggingface.co/laion/humaneness-ears-base-medium/resolve/main/release_tools/package_release.py
13.7 kB
| #!/usr/bin/env python3 | |
| """Build a credential-free, reproducible release from completed training artifacts.""" | |
| import ast | |
| import hashlib | |
| import json | |
| import shutil | |
| from collections import Counter, defaultdict | |
| from pathlib import Path | |
| HERE = Path(__file__).resolve().parent | |
| CODE = HERE.parent | |
| SCRATCH = Path('/e/scratch/reformo/schuhmann1_moss/whisper_score_regression') | |
| GEMINI = SCRATCH / 'gemini_finetune_preparation_20261005' | |
| OLD = SCRATCH / 'hf_layered_release/model' | |
| STUDY = SCRATCH / 'embedding_probe_study_20261006' | |
| OUT = SCRATCH / 'humaneness_ears_release_20261007/model' | |
| REPO = 'laion/humaneness-ears-base-medium' | |
| BENCHMARK = 'https://laion-whisper-base-small-emotion-voice-burst.static.hf.space/benchmark.html' | |
| def read(path): | |
| return json.loads(Path(path).read_text()) | |
| def write(path, value): | |
| path = Path(path) | |
| path.parent.mkdir(parents=True, exist_ok=True) | |
| path.write_text(json.dumps(value, indent=2, ensure_ascii=False, allow_nan=False) + '\n') | |
| def copy(source, relative): | |
| destination = OUT / relative | |
| destination.parent.mkdir(parents=True, exist_ok=True) | |
| if destination.suffix != '.pt' or not destination.exists() or destination.stat().st_size != Path(source).stat().st_size: | |
| shutil.copy2(source, destination) | |
| def dependency_closure(names): | |
| pending, seen = list(names), set() | |
| while pending: | |
| name = pending.pop() | |
| path = CODE / (name + '.py') | |
| if name in seen or not path.exists(): | |
| continue | |
| seen.add(name) | |
| for node in ast.walk(ast.parse(path.read_text())): | |
| if isinstance(node, ast.Import): | |
| modules = [n.name.split('.')[0] for n in node.names] | |
| elif isinstance(node, ast.ImportFrom): | |
| modules = [(node.module or '').split('.')[0]] | |
| else: | |
| continue | |
| pending.extend(m for m in modules if (CODE / (m + '.py')).exists()) | |
| return seen | |
| def stage(): | |
| OUT.mkdir(parents=True, exist_ok=True) | |
| for filename in ('LICENSE', 'classes.json', 'display_normalization.json'): | |
| copy(OLD / filename, filename) | |
| copy(GEMINI / 'prepared/checkpoint_normalization.json', 'training_normalization.json') | |
| copy(GEMINI / 'prepared/gemini_train_statistics.json', 'gemini_train_statistics.json') | |
| copy(GEMINI / 'prepared/gemini_burst_mapping.json', 'gemini_burst_mapping.json') | |
| copy(CODE / 'layered_multitask_model.py', 'model.py') | |
| shutil.copy2(HERE / 'inference.py', OUT / 'inference.py') | |
| copy(OLD / 'requirements.txt', 'requirements.txt') | |
| (OUT / 'requirements-training.txt').write_text('-r requirements.txt\npyarrow>=20\n') | |
| for size in ('base', 'small'): | |
| release_size = 'medium' if size == 'small' else size | |
| configuration = read(OLD / size / 'config.json') | |
| configuration.update(architectures=['LayeredMultiTaskWhisper'], is_encoder_decoder=False, | |
| use_cache=False, _name_or_path=REPO + '/' + release_size) | |
| write(OUT / release_size / 'config.json', configuration) | |
| copy(OLD / size / 'preprocessor_config.json', release_size + '/preprocessor_config.json') | |
| train = GEMINI / 'training' / ('whisper_' + size) | |
| copy(train / 'best.pt', release_size + '/checkpoint.pt') | |
| copy(train / 'config.json', 'training/configs/' + size + '_recorded_run.json') | |
| copy(train / 'metrics.jsonl', 'training/logs/' + size + '_epochs.jsonl') | |
| copy(train / 'COMPLETE.json', 'training/logs/' + size + '_COMPLETE.json') | |
| for kind in ['gemini_test_metrics', 'validation_metrics', 'public_metrics', | |
| 'legacy_on_flash_test_metrics', 'legacy_p3_test_metrics', 'legacy_p3_validation_metrics', | |
| 'gemini_p3_test_metrics', 'gemini_p3_validation_metrics', | |
| 'crema_matched_adapter', 'ravdess_matched_adapter']: | |
| copy(STUDY / 'whisper/gemini' / size / (kind + '.json'), 'evaluation/' + size + '/' + kind + '.json') | |
| copy(OLD / 'training' / (size + '_run_config.json'), 'training/original_curriculum/' + size + '_run_config.json') | |
| names = dependency_closure(['train_layered_curriculum', 'full_ladder_dataset', 'layered_multitask_model']) | |
| for name in sorted(names): | |
| copy(CODE / (name + '.py'), 'training/' + name + '.py') | |
| for name in ['common.py', 'data.py', 'train_whisper.py', 'train_whisper.sbatch', 'prepare.py', 'backfill.py', 'configure.py']: | |
| copy(CODE / 'gemini_finetune' / name, 'training/gemini_finetune/' + name) | |
| for name in ['annotate.py', 'gemini_voice_annotation_master_prompt_detailed.txt']: | |
| copy(CODE / 'gemini_s10_smoke' / name, 'training/gemini_s10_smoke/' + name) | |
| teachers = CODE.parent / 'vocal_burst_pool/p3' | |
| for name in ['annotate_dnsmos.py', 'annotate_audiobox.py', 'annotate_voiceclap.py', 'annotate_empathic.py']: | |
| copy(teachers / name, 'vocal_burst_pool/p3/' + name) | |
| for name in dependency_closure(['evaluate_public_benchmarks']): | |
| copy(CODE / (name + '.py'), 'training/' + name + '.py') | |
| for path in sorted((CODE / 'embedding_probe_study').glob('*.py')): | |
| copy(path, 'training/embedding_probe_study/' + path.name) | |
| helper = Path('/e/scratch/reformo/schuhmann1_moss/out/m2_600m_ladder/code_v4_highlr_20260920/tar_member_map.py') | |
| if helper.exists(): | |
| copy(helper, 'training/tar_member_map.py') | |
| copy(SCRATCH.parent / 'code/fastgen.sh', 'training/original_curriculum/fastgen.sh') | |
| copy(HERE / 'make_finetune_config.py', 'training/make_finetune_config.py') | |
| for name in ['export_and_verify.py', 'export_rank.py', 'verify.sbatch', 'package_release.py', 'publish_release.py']: | |
| copy(HERE / name, 'release_tools/' + name) | |
| copy(SCRATCH / 'gemini_multisource_100h_20261004/combined_summary.json', 'training/provenance/selected_pool.json') | |
| copy(GEMINI / 'prepared/TRAINING_READY.json', 'training/provenance/TRAINING_READY.json') | |
| (OUT / 'NOTICE.md').write_text('''# License and attribution | |
| Author: **Christoph Schuhmann**. Organization: **LAION**. Release: 7 October 2026. | |
| LAION's new fine-tuned weights, code and documentation: **CC BY 4.0**, see `LICENSE`. | |
| Please credit Christoph Schuhmann and LAION, link this repository and indicate changes. | |
| OpenAI Whisper's original code and weights retain their MIT notice: | |
| https://github.com/openai/whisper/blob/main/LICENSE | |
| Hugging Face Transformers is an external Apache-2.0 dependency: | |
| https://github.com/huggingface/transformers/blob/main/LICENSE | |
| Teacher models and source audio retain their own terms; they are not redistributed here. | |
| Orange's identity teacher card declares CC BY-SA 3.0: | |
| https://huggingface.co/Orange/Speaker-wavLM-id | |
| Orange timbre: https://huggingface.co/Orange/Speaker-wavLM-tbr | |
| The release license is not a grant of rights over upstream datasets or packages. | |
| ''') | |
| (OUT / '.gitattributes').write_text('*.safetensors filter=lfs diff=lfs merge=lfs -text\n*.pt filter=lfs diff=lfs merge=lfs -text\n') | |
| print('STAGED', str(OUT), 'training dependency modules', len(names), flush=True) | |
| def data_summary(): | |
| summary = OUT / 'training/provenance/valid_split_summary.json' | |
| if summary.exists(): | |
| return read(summary) | |
| counts, hours, languages, sources = Counter(), defaultdict(float), Counter(), Counter() | |
| fingerprint = hashlib.sha256() | |
| with (GEMINI / 'prepared/targets.jsonl').open('rb') as stream: | |
| for raw in stream: | |
| fingerprint.update(raw) | |
| row = json.loads(raw) | |
| if not row['ready_for_whisper_training']: | |
| continue | |
| counts[row['split']] += 1 | |
| hours[row['split']] += row['duration_s'] / 3600 | |
| languages[row['annotation']['language']] += 1 | |
| for source in {m['source'] for m in row['source_memberships']}: | |
| sources[source] += 1 | |
| assert counts == {'train': 58380, 'validation': 3239, 'test': 3544}, counts | |
| result = {'valid_clips': sum(counts.values()), 'split_clips': dict(counts), 'split_hours': dict(hours), | |
| 'valid_total_hours': sum(hours.values()), 'language_strings': dict(languages), | |
| 'source_membership_counts_nonexclusive': dict(sources), | |
| 'prepared_targets_sha256': fingerprint.hexdigest(), | |
| 'deduplication': 'Exact selected compressed audio SHA256 across all selected sources/tasks', | |
| 'split': '90/5/5 deterministic group hash; explicit speaker or synthetic family before exact audio SHA', | |
| 'excluded_clips': 1036, 'captions_used_as_model_inputs': False} | |
| write(summary, result) | |
| return result | |
| def card(): | |
| data = data_summary() | |
| template = (HERE / 'README_template.md').read_text() | |
| base = read(OUT / 'evaluation/base/gemini_test_metrics.json') | |
| small = read(OUT / 'evaluation/small/gemini_test_metrics.json') | |
| def num(v): | |
| return 'Unavailable' if v is None else f'{v:.4f}' | |
| def score(m, name, key): | |
| return next(x[key] for x in m['per_score'] if x['score'] == name) | |
| def family(m, prefix): | |
| values = [x['normalized_mae'] for x in m['per_score'] if x['score'].startswith(prefix) and x['normalized_mae'] is not None] | |
| return sum(values) / len(values) | |
| rows = [] | |
| for task, function in [ | |
| ('Burst frame F1 β', lambda m: m['frame_f1']), | |
| ('Burst event F1, IoU β₯0.5 β', lambda m: m['burst_localization']['0.5']['f1']), | |
| ('Burst start MAE on IoU β₯0.1 matches, seconds β', lambda m: m['burst_start_mae_s']), | |
| ('Burst end MAE on IoU β₯0.1 matches, seconds β', lambda m: m['burst_end_mae_s']), | |
| ('53-class accuracy with reference spans β', lambda m: m['burst_class_oracle_span_accuracy']), | |
| ('53-class accuracy on matched predicted spans β', lambda m: m['burst_matched_event_class_accuracy']), | |
| ('Orange timbre cosine β (1,718 valid single-speaker clips)', lambda m: m['speaker_cosine']['timbre']['mean']), | |
| ('Orange identity cosine β (1,718 valid single-speaker clips)', lambda m: m['speaker_cosine']['identity']['mean']), | |
| ('CPS raw MAE, characters/second β', lambda m: m['cps_raw_mae']), | |
| ('Genuineness raw MAE on 0β6 scale β (3,544 clips)', lambda m: score(m, 'genuineness_0_6', 'raw_mae')), | |
| ('Genuineness normalized MAE β', lambda m: score(m, 'genuineness_0_6', 'normalized_mae')), | |
| ('Blend raw MAE on 0β10 scale β (2,391 valid clips)', lambda m: score(m, 'blend_0_10', 'raw_mae')), | |
| ('Blend normalized MAE β', lambda m: score(m, 'blend_0_10', 'normalized_mae')), | |
| ('40 emotions, mean normalized MAE β', lambda m: family(m, 'emo_')), | |
| ('57 VoiceNet dimensions, mean normalized MAE β', lambda m: family(m, 'vn_')), | |
| ('AudioBox four axes, mean normalized MAE β', lambda m: family(m, 'audiobox:')), | |
| ('DNSMOS seven outputs, mean normalized MAE β', lambda m: family(m, 'dnsmos:'))]: | |
| rows.append('| ' + task + ' | ' + num(function(base)) + ' | ' + num(function(small)) + ' |') | |
| holdout_table = '| Flash test target / metric | Base | Medium |\n| --- | ---: | ---: |\n' + '\n'.join(rows) | |
| public_rows = [] | |
| for task, key, metric in [('EmoNet-Voice intensity, Pearson r β', 'emonet', 'pearson'), | |
| ('EmoNet-Voice intensity, Spearman Ο β', 'emonet', 'spearman'), | |
| ('VoiceNet-Emo, mean prompt Spearman Ο β', 'emolia-emo', 'mean_prompt_spearman'), | |
| ('VoiceNet-Ext / emolia-dim, mean prompt Spearman Ο β', 'emolia-dim', 'mean_prompt_spearman')]: | |
| values = [] | |
| for size in ['base', 'small']: | |
| result = read(OUT / 'evaluation' / size / 'public_metrics.json')[key] | |
| values.append(result['all_mapped_40' if key == 'emonet' else 'repo_min_2_raters'][metric]) | |
| public_rows.append('| ' + task + ' | ' + num(values[0]) + ' | ' + num(values[1]) + ' |') | |
| for key in ['crema', 'ravdess']: | |
| values = [read(OUT / 'evaluation' / s / (key + '_matched_adapter.json'))['metrics']['accuracy'] for s in ['base', 'small']] | |
| public_rows.append('| ' + ('CREMA-D' if key == 'crema' else 'RAVDESS') + ' matched actor-CV score-MLP accuracy β | ' + num(values[0]) + ' | ' + num(values[1]) + ' |') | |
| public_table = '| Human-label benchmark / metric | Base | Medium |\n| --- | ---: | ---: |\n' + '\n'.join(public_rows) | |
| target_rows = [] | |
| for index, (b, s) in enumerate(zip(base['per_score'], small['per_score'])): | |
| assert b['score'] == s['score'] | |
| target_rows.append('| ' + str(index) + ' | `' + b['score'] + '` | ' + str(b['n']) + ' | ' + ' | '.join( | |
| num(m[k]) for m in [b, s] for k in ['raw_mae', 'normalized_mae', 'pearson', 'spearman']) + ' |') | |
| target_table = '| Index | Exact target key | Valid N | Base raw MAE | Base norm. MAE | Base r | Base Ο | Medium raw MAE | Medium norm. MAE | Medium r | Medium Ο |\n| ---: | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |\n' + '\n'.join(target_rows) | |
| names = read(OUT / 'classes.json')['names'] | |
| classes = '| ID | Canonical vocal-burst class |\n| ---: | --- |\n' + '\n'.join('| ' + str(i) + ' | ' + name + ' |' for i, name in enumerate(names)) | |
| card = template.replace('{{HOLDOUT_TABLE}}', holdout_table).replace('{{PUBLIC_TABLE}}', public_table).replace('{{TARGET_TABLE}}', target_table).replace('{{CLASS_TABLE}}', classes).replace('{{VALID_HOURS}}', f"{data['valid_total_hours']:.2f}") | |
| assert '{{' not in card.replace('author = {Schuhmann, Christoph and {LAION}}', ''), 'Unexpanded placeholder' | |
| (OUT / 'README.md').write_text(card) | |
| print('WROTE model card', len(card.encode()), flush=True) | |
| if __name__ == '__main__': | |
| stage() | |
| card() | |