aether-phase1-build / scripts /curate_slivers.py
SupremeD's picture
Upload folder using huggingface_hub
bed4b7b verified
Raw History Blame Contribute Delete
4.02 kB
#!/usr/bin/env python
"""
Aether Phase-1 sliver curation. Coverage-balanced (NOT random), Tier-A sources only,
every item caption+tagged (alignment needs paired text). Writes /work/slivers/{vision,audio,3d}/
as JSONL + a sha256 provenance manifest.
Caption/tag stack (all apache/MIT — keeps slivers Tier-A):
captions: Gliese-Qwen3.5-9B-Abliterated-Caption (VLM) | JoyCaption-Q4 (6900XT) | CapRL
tags: prithiv SigLIP2 classifiers (scene/material/age/gender/object) | WD-tagger
3D caption: render->multi-view->caption the object + geometry/asset-type tags.
"""
import os, json, hashlib, random
random.seed(0)
OUT = "/work/slivers"
# ---------- COMPOSITION (ratios per review; sources Tier-A-corrected) ----------
VISION = { # ~500K-1M
"natural": (0.40, ["mvp-lab/LLaVA-OneVision-1.5-Mid-Training-85M"]), # has captions
"doc_ocr": (0.30, ["Salesforce/blip3-ocr-200m", "prithivMLmods/Latex-KIE", "SYNTH:SynthDoG"]),
"ui_web": (0.20, ["MBZUAI/Web2Code", "zai-org/Vision2Web"]), # +bbox
"dense_technical":(0.10, ["LOCAL:electrical_plans"]), # our plans -> micro-scale
}
AUDIO = { # ~500K (acoustic diversity, not speech-only)
"clean_speech": (0.60, ["facebook/voxpopuli"]), # CC0, transcripts=text
"noisy_conv": (0.20, ["CoVoST/Common-Voice-CC0"]), # CC0 (NOT LibriSpeech=CC-BY)
"sound_events": (0.20, ["Freesound-CC0-subset", "SYNTH:dsp_events"]), # NOT AudioSet(YouTube)
}
THREED = { # ~300K-500K, PAIRED: [multiview+poses] <-> [SLAT] <-> [text]; balance topo density
"primitives": (0.30, ["SAGE-10k:primitive"]), # box/cylinder
"props": (0.40, ["SAGE-10k:prop", "TRELLIS-500K", "OpenGameArt-CC0:prop"]), # tools/mech
"organic": (0.30, ["OpenGameArt-CC0:char", "SAGE-10k:organic"]), # characters/creatures
}
TARGET = {"vision": 800_000, "audio": 500_000, "3d": 400_000}
def sha256(b): return hashlib.sha256(b).hexdigest()
def ensure_caption_tag(item, modality):
"""Every item MUST carry caption + tags. Native caption if present+quality; else generate clean."""
if not item.get("caption"):
item["caption"] = caption_clean(item, modality) # Gliese/JoyCaption (VLM) or render->caption for 3D
if not item.get("tags"):
item["tags"] = tag_clean(item, modality) # SigLIP2 classifiers / WD-tagger / geom+asset-type
item["caption_src"] = item.get("caption_src", "native_or_clean_generated")
return item
def caption_clean(item, modality): ... # TODO: wire Gliese-Qwen3.5-Abliterated-Caption (rental) / JoyCaption (6900XT)
def tag_clean(item, modality): ... # TODO: wire prithiv SigLIP2 classifiers (offline labelers)
def render_multiview(mesh_path, n=6): ...# 3D: glb/mesh -> N views + camera poses (blender/kaolin)
def encode_slat(mesh_path): ... # 3D: TRELLIS SLAT latents (from staged TRELLIS encoder)
def curate(modality, spec, target):
os.makedirs(f"{OUT}/{modality}", exist_ok=True)
manifest, n_written = [], 0
for bucket, (ratio, sources) in spec.items():
quota = int(target * ratio)
# stream each source, dedup, cap at quota/len(sources), caption+tag, write
# (streaming impl per-source: datasets.load_dataset(..., streaming=True) for HF ids;
# LOCAL:/SYNTH:/render for our-own; 3D buckets pair multiview+SLAT+text)
...
manifest.append({"bucket": bucket, "ratio": ratio, "sources": sources, "quota": quota})
json.dump({"modality": modality, "target": target, "buckets": manifest},
open(f"{OUT}/{modality}/manifest.json", "w"), indent=2)
print(f"{modality}: target {target}, buckets {list(spec)} -> manifest written")
if __name__ == "__main__":
for m, spec in [("vision", VISION), ("audio", AUDIO), ("3d", THREED)]:
curate(m, spec, TARGET[m])
print("sliver curation scaffold ready — wire caption/tag + streaming loaders, then run on rental.")