#!/usr/bin/env python """ Aether Phase-1 sliver curation. Coverage-balanced (NOT random), Tier-A sources only, every item caption+tagged (alignment needs paired text). Writes /work/slivers/{vision,audio,3d}/ as JSONL + a sha256 provenance manifest. Caption/tag stack (all apache/MIT — keeps slivers Tier-A): captions: Gliese-Qwen3.5-9B-Abliterated-Caption (VLM) | JoyCaption-Q4 (6900XT) | CapRL tags: prithiv SigLIP2 classifiers (scene/material/age/gender/object) | WD-tagger 3D caption: render->multi-view->caption the object + geometry/asset-type tags. """ import os, json, hashlib, random random.seed(0) OUT = "/work/slivers" # ---------- COMPOSITION (ratios per review; sources Tier-A-corrected) ---------- VISION = { # ~500K-1M "natural": (0.40, ["mvp-lab/LLaVA-OneVision-1.5-Mid-Training-85M"]), # has captions "doc_ocr": (0.30, ["Salesforce/blip3-ocr-200m", "prithivMLmods/Latex-KIE", "SYNTH:SynthDoG"]), "ui_web": (0.20, ["MBZUAI/Web2Code", "zai-org/Vision2Web"]), # +bbox "dense_technical":(0.10, ["LOCAL:electrical_plans"]), # our plans -> micro-scale } AUDIO = { # ~500K (acoustic diversity, not speech-only) "clean_speech": (0.60, ["facebook/voxpopuli"]), # CC0, transcripts=text "noisy_conv": (0.20, ["CoVoST/Common-Voice-CC0"]), # CC0 (NOT LibriSpeech=CC-BY) "sound_events": (0.20, ["Freesound-CC0-subset", "SYNTH:dsp_events"]), # NOT AudioSet(YouTube) } THREED = { # ~300K-500K, PAIRED: [multiview+poses] <-> [SLAT] <-> [text]; balance topo density "primitives": (0.30, ["SAGE-10k:primitive"]), # box/cylinder "props": (0.40, ["SAGE-10k:prop", "TRELLIS-500K", "OpenGameArt-CC0:prop"]), # tools/mech "organic": (0.30, ["OpenGameArt-CC0:char", "SAGE-10k:organic"]), # characters/creatures } TARGET = {"vision": 800_000, "audio": 500_000, "3d": 400_000} def sha256(b): return hashlib.sha256(b).hexdigest() def ensure_caption_tag(item, modality): """Every item MUST carry caption + tags. Native caption if present+quality; else generate clean.""" if not item.get("caption"): item["caption"] = caption_clean(item, modality) # Gliese/JoyCaption (VLM) or render->caption for 3D if not item.get("tags"): item["tags"] = tag_clean(item, modality) # SigLIP2 classifiers / WD-tagger / geom+asset-type item["caption_src"] = item.get("caption_src", "native_or_clean_generated") return item def caption_clean(item, modality): ... # TODO: wire Gliese-Qwen3.5-Abliterated-Caption (rental) / JoyCaption (6900XT) def tag_clean(item, modality): ... # TODO: wire prithiv SigLIP2 classifiers (offline labelers) def render_multiview(mesh_path, n=6): ...# 3D: glb/mesh -> N views + camera poses (blender/kaolin) def encode_slat(mesh_path): ... # 3D: TRELLIS SLAT latents (from staged TRELLIS encoder) def curate(modality, spec, target): os.makedirs(f"{OUT}/{modality}", exist_ok=True) manifest, n_written = [], 0 for bucket, (ratio, sources) in spec.items(): quota = int(target * ratio) # stream each source, dedup, cap at quota/len(sources), caption+tag, write # (streaming impl per-source: datasets.load_dataset(..., streaming=True) for HF ids; # LOCAL:/SYNTH:/render for our-own; 3D buckets pair multiview+SLAT+text) ... manifest.append({"bucket": bucket, "ratio": ratio, "sources": sources, "quota": quota}) json.dump({"modality": modality, "target": target, "buckets": manifest}, open(f"{OUT}/{modality}/manifest.json", "w"), indent=2) print(f"{modality}: target {target}, buckets {list(spec)} -> manifest written") if __name__ == "__main__": for m, spec in [("vision", VISION), ("audio", AUDIO), ("3d", THREED)]: curate(m, spec, TARGET[m]) print("sliver curation scaffold ready — wire caption/tag + streaming loaders, then run on rental.")