ChristophSchuhmann's picture
Align release methods and republishing code with Humaneness Ears
1797421 verified
Raw History Blame Contribute Delete
13.7 kB
#!/usr/bin/env python3
"""Build a credential-free, reproducible release from completed training artifacts."""
import ast
import hashlib
import json
import shutil
from collections import Counter, defaultdict
from pathlib import Path
HERE = Path(__file__).resolve().parent
CODE = HERE.parent
SCRATCH = Path('/e/scratch/reformo/schuhmann1_moss/whisper_score_regression')
GEMINI = SCRATCH / 'gemini_finetune_preparation_20261005'
OLD = SCRATCH / 'hf_layered_release/model'
STUDY = SCRATCH / 'embedding_probe_study_20261006'
OUT = SCRATCH / 'humaneness_ears_release_20261007/model'
REPO = 'laion/humaneness-ears-base-medium'
BENCHMARK = 'https://laion-whisper-base-small-emotion-voice-burst.static.hf.space/benchmark.html'
def read(path):
return json.loads(Path(path).read_text())
def write(path, value):
path = Path(path)
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(value, indent=2, ensure_ascii=False, allow_nan=False) + '\n')
def copy(source, relative):
destination = OUT / relative
destination.parent.mkdir(parents=True, exist_ok=True)
if destination.suffix != '.pt' or not destination.exists() or destination.stat().st_size != Path(source).stat().st_size:
shutil.copy2(source, destination)
def dependency_closure(names):
pending, seen = list(names), set()
while pending:
name = pending.pop()
path = CODE / (name + '.py')
if name in seen or not path.exists():
continue
seen.add(name)
for node in ast.walk(ast.parse(path.read_text())):
if isinstance(node, ast.Import):
modules = [n.name.split('.')[0] for n in node.names]
elif isinstance(node, ast.ImportFrom):
modules = [(node.module or '').split('.')[0]]
else:
continue
pending.extend(m for m in modules if (CODE / (m + '.py')).exists())
return seen
def stage():
OUT.mkdir(parents=True, exist_ok=True)
for filename in ('LICENSE', 'classes.json', 'display_normalization.json'):
copy(OLD / filename, filename)
copy(GEMINI / 'prepared/checkpoint_normalization.json', 'training_normalization.json')
copy(GEMINI / 'prepared/gemini_train_statistics.json', 'gemini_train_statistics.json')
copy(GEMINI / 'prepared/gemini_burst_mapping.json', 'gemini_burst_mapping.json')
copy(CODE / 'layered_multitask_model.py', 'model.py')
shutil.copy2(HERE / 'inference.py', OUT / 'inference.py')
copy(OLD / 'requirements.txt', 'requirements.txt')
(OUT / 'requirements-training.txt').write_text('-r requirements.txt\npyarrow>=20\n')
for size in ('base', 'small'):
release_size = 'medium' if size == 'small' else size
configuration = read(OLD / size / 'config.json')
configuration.update(architectures=['LayeredMultiTaskWhisper'], is_encoder_decoder=False,
use_cache=False, _name_or_path=REPO + '/' + release_size)
write(OUT / release_size / 'config.json', configuration)
copy(OLD / size / 'preprocessor_config.json', release_size + '/preprocessor_config.json')
train = GEMINI / 'training' / ('whisper_' + size)
copy(train / 'best.pt', release_size + '/checkpoint.pt')
copy(train / 'config.json', 'training/configs/' + size + '_recorded_run.json')
copy(train / 'metrics.jsonl', 'training/logs/' + size + '_epochs.jsonl')
copy(train / 'COMPLETE.json', 'training/logs/' + size + '_COMPLETE.json')
for kind in ['gemini_test_metrics', 'validation_metrics', 'public_metrics',
'legacy_on_flash_test_metrics', 'legacy_p3_test_metrics', 'legacy_p3_validation_metrics',
'gemini_p3_test_metrics', 'gemini_p3_validation_metrics',
'crema_matched_adapter', 'ravdess_matched_adapter']:
copy(STUDY / 'whisper/gemini' / size / (kind + '.json'), 'evaluation/' + size + '/' + kind + '.json')
copy(OLD / 'training' / (size + '_run_config.json'), 'training/original_curriculum/' + size + '_run_config.json')
names = dependency_closure(['train_layered_curriculum', 'full_ladder_dataset', 'layered_multitask_model'])
for name in sorted(names):
copy(CODE / (name + '.py'), 'training/' + name + '.py')
for name in ['common.py', 'data.py', 'train_whisper.py', 'train_whisper.sbatch', 'prepare.py', 'backfill.py', 'configure.py']:
copy(CODE / 'gemini_finetune' / name, 'training/gemini_finetune/' + name)
for name in ['annotate.py', 'gemini_voice_annotation_master_prompt_detailed.txt']:
copy(CODE / 'gemini_s10_smoke' / name, 'training/gemini_s10_smoke/' + name)
teachers = CODE.parent / 'vocal_burst_pool/p3'
for name in ['annotate_dnsmos.py', 'annotate_audiobox.py', 'annotate_voiceclap.py', 'annotate_empathic.py']:
copy(teachers / name, 'vocal_burst_pool/p3/' + name)
for name in dependency_closure(['evaluate_public_benchmarks']):
copy(CODE / (name + '.py'), 'training/' + name + '.py')
for path in sorted((CODE / 'embedding_probe_study').glob('*.py')):
copy(path, 'training/embedding_probe_study/' + path.name)
helper = Path('/e/scratch/reformo/schuhmann1_moss/out/m2_600m_ladder/code_v4_highlr_20260920/tar_member_map.py')
if helper.exists():
copy(helper, 'training/tar_member_map.py')
copy(SCRATCH.parent / 'code/fastgen.sh', 'training/original_curriculum/fastgen.sh')
copy(HERE / 'make_finetune_config.py', 'training/make_finetune_config.py')
for name in ['export_and_verify.py', 'export_rank.py', 'verify.sbatch', 'package_release.py', 'publish_release.py']:
copy(HERE / name, 'release_tools/' + name)
copy(SCRATCH / 'gemini_multisource_100h_20261004/combined_summary.json', 'training/provenance/selected_pool.json')
copy(GEMINI / 'prepared/TRAINING_READY.json', 'training/provenance/TRAINING_READY.json')
(OUT / 'NOTICE.md').write_text('''# License and attribution
Author: **Christoph Schuhmann**. Organization: **LAION**. Release: 7 October 2026.
LAION's new fine-tuned weights, code and documentation: **CC BY 4.0**, see `LICENSE`.
Please credit Christoph Schuhmann and LAION, link this repository and indicate changes.
OpenAI Whisper's original code and weights retain their MIT notice:
https://github.com/openai/whisper/blob/main/LICENSE
Hugging Face Transformers is an external Apache-2.0 dependency:
https://github.com/huggingface/transformers/blob/main/LICENSE
Teacher models and source audio retain their own terms; they are not redistributed here.
Orange's identity teacher card declares CC BY-SA 3.0:
https://huggingface.co/Orange/Speaker-wavLM-id
Orange timbre: https://huggingface.co/Orange/Speaker-wavLM-tbr
The release license is not a grant of rights over upstream datasets or packages.
''')
(OUT / '.gitattributes').write_text('*.safetensors filter=lfs diff=lfs merge=lfs -text\n*.pt filter=lfs diff=lfs merge=lfs -text\n')
print('STAGED', str(OUT), 'training dependency modules', len(names), flush=True)
def data_summary():
summary = OUT / 'training/provenance/valid_split_summary.json'
if summary.exists():
return read(summary)
counts, hours, languages, sources = Counter(), defaultdict(float), Counter(), Counter()
fingerprint = hashlib.sha256()
with (GEMINI / 'prepared/targets.jsonl').open('rb') as stream:
for raw in stream:
fingerprint.update(raw)
row = json.loads(raw)
if not row['ready_for_whisper_training']:
continue
counts[row['split']] += 1
hours[row['split']] += row['duration_s'] / 3600
languages[row['annotation']['language']] += 1
for source in {m['source'] for m in row['source_memberships']}:
sources[source] += 1
assert counts == {'train': 58380, 'validation': 3239, 'test': 3544}, counts
result = {'valid_clips': sum(counts.values()), 'split_clips': dict(counts), 'split_hours': dict(hours),
'valid_total_hours': sum(hours.values()), 'language_strings': dict(languages),
'source_membership_counts_nonexclusive': dict(sources),
'prepared_targets_sha256': fingerprint.hexdigest(),
'deduplication': 'Exact selected compressed audio SHA256 across all selected sources/tasks',
'split': '90/5/5 deterministic group hash; explicit speaker or synthetic family before exact audio SHA',
'excluded_clips': 1036, 'captions_used_as_model_inputs': False}
write(summary, result)
return result
def card():
data = data_summary()
template = (HERE / 'README_template.md').read_text()
base = read(OUT / 'evaluation/base/gemini_test_metrics.json')
small = read(OUT / 'evaluation/small/gemini_test_metrics.json')
def num(v):
return 'Unavailable' if v is None else f'{v:.4f}'
def score(m, name, key):
return next(x[key] for x in m['per_score'] if x['score'] == name)
def family(m, prefix):
values = [x['normalized_mae'] for x in m['per_score'] if x['score'].startswith(prefix) and x['normalized_mae'] is not None]
return sum(values) / len(values)
rows = []
for task, function in [
('Burst frame F1 ↑', lambda m: m['frame_f1']),
('Burst event F1, IoU β‰₯0.5 ↑', lambda m: m['burst_localization']['0.5']['f1']),
('Burst start MAE on IoU β‰₯0.1 matches, seconds ↓', lambda m: m['burst_start_mae_s']),
('Burst end MAE on IoU β‰₯0.1 matches, seconds ↓', lambda m: m['burst_end_mae_s']),
('53-class accuracy with reference spans ↑', lambda m: m['burst_class_oracle_span_accuracy']),
('53-class accuracy on matched predicted spans ↑', lambda m: m['burst_matched_event_class_accuracy']),
('Orange timbre cosine ↑ (1,718 valid single-speaker clips)', lambda m: m['speaker_cosine']['timbre']['mean']),
('Orange identity cosine ↑ (1,718 valid single-speaker clips)', lambda m: m['speaker_cosine']['identity']['mean']),
('CPS raw MAE, characters/second ↓', lambda m: m['cps_raw_mae']),
('Genuineness raw MAE on 0–6 scale ↓ (3,544 clips)', lambda m: score(m, 'genuineness_0_6', 'raw_mae')),
('Genuineness normalized MAE ↓', lambda m: score(m, 'genuineness_0_6', 'normalized_mae')),
('Blend raw MAE on 0–10 scale ↓ (2,391 valid clips)', lambda m: score(m, 'blend_0_10', 'raw_mae')),
('Blend normalized MAE ↓', lambda m: score(m, 'blend_0_10', 'normalized_mae')),
('40 emotions, mean normalized MAE ↓', lambda m: family(m, 'emo_')),
('57 VoiceNet dimensions, mean normalized MAE ↓', lambda m: family(m, 'vn_')),
('AudioBox four axes, mean normalized MAE ↓', lambda m: family(m, 'audiobox:')),
('DNSMOS seven outputs, mean normalized MAE ↓', lambda m: family(m, 'dnsmos:'))]:
rows.append('| ' + task + ' | ' + num(function(base)) + ' | ' + num(function(small)) + ' |')
holdout_table = '| Flash test target / metric | Base | Medium |\n| --- | ---: | ---: |\n' + '\n'.join(rows)
public_rows = []
for task, key, metric in [('EmoNet-Voice intensity, Pearson r ↑', 'emonet', 'pearson'),
('EmoNet-Voice intensity, Spearman ρ ↑', 'emonet', 'spearman'),
('VoiceNet-Emo, mean prompt Spearman ρ ↑', 'emolia-emo', 'mean_prompt_spearman'),
('VoiceNet-Ext / emolia-dim, mean prompt Spearman ρ ↑', 'emolia-dim', 'mean_prompt_spearman')]:
values = []
for size in ['base', 'small']:
result = read(OUT / 'evaluation' / size / 'public_metrics.json')[key]
values.append(result['all_mapped_40' if key == 'emonet' else 'repo_min_2_raters'][metric])
public_rows.append('| ' + task + ' | ' + num(values[0]) + ' | ' + num(values[1]) + ' |')
for key in ['crema', 'ravdess']:
values = [read(OUT / 'evaluation' / s / (key + '_matched_adapter.json'))['metrics']['accuracy'] for s in ['base', 'small']]
public_rows.append('| ' + ('CREMA-D' if key == 'crema' else 'RAVDESS') + ' matched actor-CV score-MLP accuracy ↑ | ' + num(values[0]) + ' | ' + num(values[1]) + ' |')
public_table = '| Human-label benchmark / metric | Base | Medium |\n| --- | ---: | ---: |\n' + '\n'.join(public_rows)
target_rows = []
for index, (b, s) in enumerate(zip(base['per_score'], small['per_score'])):
assert b['score'] == s['score']
target_rows.append('| ' + str(index) + ' | `' + b['score'] + '` | ' + str(b['n']) + ' | ' + ' | '.join(
num(m[k]) for m in [b, s] for k in ['raw_mae', 'normalized_mae', 'pearson', 'spearman']) + ' |')
target_table = '| Index | Exact target key | Valid N | Base raw MAE | Base norm. MAE | Base r | Base ρ | Medium raw MAE | Medium norm. MAE | Medium r | Medium ρ |\n| ---: | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |\n' + '\n'.join(target_rows)
names = read(OUT / 'classes.json')['names']
classes = '| ID | Canonical vocal-burst class |\n| ---: | --- |\n' + '\n'.join('| ' + str(i) + ' | ' + name + ' |' for i, name in enumerate(names))
card = template.replace('{{HOLDOUT_TABLE}}', holdout_table).replace('{{PUBLIC_TABLE}}', public_table).replace('{{TARGET_TABLE}}', target_table).replace('{{CLASS_TABLE}}', classes).replace('{{VALID_HOURS}}', f"{data['valid_total_hours']:.2f}")
assert '{{' not in card.replace('author = {Schuhmann, Christoph and {LAION}}', ''), 'Unexpanded placeholder'
(OUT / 'README.md').write_text(card)
print('WROTE model card', len(card.encode()), flush=True)
if __name__ == '__main__':
stage()
card()