Text-to-Speech
English
German
voice-acting
qwen3
moss-audio-tokenizer-v2
audio-generation
Humaneness-Voice-Small / code /verify_caption_curriculum.py
ChristophSchuhmann's picture
Document architecture, prompts, code, and full run statistics
5de8603 verified
Raw History Blame Contribute Delete
3.44 kB
#!/usr/bin/env python3
"""Fail closed on asset/manifests before any GPU smoke can be frozen."""
from __future__ import annotations
import hashlib
import json
import math
from pathlib import Path
import random
import numpy as np
from caption_curriculum_dataset import DESC_DTYPE, MISSING
ROOT = Path("/e/scratch/reformo/schuhmann1_moss/out/m2_600m_ladder/caption_curriculum_v1")
def sha(path):
h = hashlib.sha256()
with Path(path).open("rb") as stream:
for block in iter(lambda: stream.read(8 << 20), b""):
h.update(block)
return h.hexdigest()
def main():
assets_path = ROOT / "assets/manifest.json"
assets = json.loads(assets_path.read_text())
assert assets["status"] == "complete" and assets["annotations"]["rows"] == 3_682_649
assert assets["references"]["rows"] > 18_000_000
summaries = []
for number in range(1, 11):
root = ROOT / "manifests" / f"S{number}"
manifest = json.loads((root / "manifest.json").read_text())
assert manifest["status"] == "complete" and manifest["format"] == "m2-caption-curriculum-v1"
assert manifest["assets_manifest_sha256"] == sha(assets_path)
for name in ("descriptors", "order"):
assert manifest[name + "_sha256"] == sha(root / manifest[name])
descriptors = np.load(root / manifest["descriptors"], mmap_mode="r", allow_pickle=False)
order = np.load(root / manifest["order"], mmap_mode="r", allow_pickle=False)
assert descriptors.dtype == DESC_DTYPE
assert len(descriptors) == len(order) == manifest["presentations"]
assert int(order.min()) == 0 and int(order.max()) == len(order) - 1
family_counts = {int(k): int(v) for k, v in manifest["family_counts"].items()}
family_refs = {int(k): int(v) for k, v in manifest["family_reference_counts"].items()}
family_eligible = {int(k): int(v) for k, v in
manifest["family_reference_eligible_counts"].items()}
for family in range(6):
eligible = family_eligible.get(family, 0)
if eligible:
ratio = family_refs.get(family, 0) / eligible
tolerance = max(0.005, 4.0 / math.sqrt(eligible))
assert abs(ratio - 0.5) <= tolerance, (number, family, ratio, tolerance)
assert family_refs.get(family, 0) <= eligible <= family_counts[family]
assert family_refs[6] == family_counts[6]
assert abs(family_counts[6] / manifest["unique_audio_occurrences"] - 0.10) < 0.002
rng = random.Random(20260923 + number)
for _ in range(10_000):
row = descriptors[rng.randrange(len(descriptors))]
assert int(row["family"]) < 7 and int(row["mode"]) < 2
assert (int(row["mode"]) == 1) == (int(row["reference"]) != MISSING)
assert (int(row["family"]) in (2, 3, 4, 5)) == (int(row["annotation"]) != MISSING)
summaries.append({"stage": number, "presentations": len(descriptors),
"reference_ratio_main": sum(family_refs.get(x, 0) for x in range(6)) /
sum(family_counts.get(x, 0) for x in range(6))})
report = {"status": "PASS", "assets": str(assets_path), "stages": summaries}
(ROOT / "VERIFY.json").write_text(json.dumps(report, indent=2) + "\n")
print(json.dumps(report, indent=2))
if __name__ == "__main__":
main()