#!/usr/bin/env python """ Aether Phase-1 sliver curation — coverage-balanced, Tier-A-gated, every item caption+tagged, then pre-extracted to the portable feature cache (feeds train_cached.py). PIPELINE (per item): source stream -> TIER-A GATE (assert license) -> ensure caption+tag -> encode with the frozen tower -> write cache//.pt + append index.jsonl -> sha256 provenance manifest. Tier-A bar (feedback_tier_a_no_citation_bar): apache-2.0 / MIT / CC0 / public-domain / OpenMDW-1.1 / CDLA-Permissive-2.0 ONLY. Reject CC-BY(any)/ODC-BY/SA/NC/GPL/OpenRAIL/other/gated. Re-gate EVERY run (feedback_preflight_tiera_gate). Caption/tag stack (all apache/MIT — keeps slivers Tier-A): captions: JoyCaption-Q4 (6900XT) | CapRL-method | native field when the source carries one tags: prithiv SigLIP2 classifiers (scene/material/object) | WD-tagger 3D: render -> multi-view -> caption the object + geometry/asset-type tags. Run on a box with the frozen tower staged (MI300 for scale). Sources come from ~/AETHER_DATASET_MASTER.md + tiera_catalog/catalog_classified.csv (already gated). """ import os, sys, json, hashlib sys.path.insert(0, "/work") TIER_A = {"apache-2.0","apache2.0","apache","mit","cc0-1.0","cc0","public-domain","pd", "openmdw-1.1","openmdw","cdla-permissive-2.0","cdla-permissive"} CACHE = "/work/slivers_cache" # ---------- COMPOSITION (ratios per review; sources Tier-A-gated) ---------- VISION = { # ~800K "natural": (0.40, [""]), "doc_ocr": (0.30, ["SYNTH:SynthDoG", "LOCAL:electrical_plans_md"]), "ui_web": (0.20, [""]), "dense_technical": (0.10, ["LOCAL:electrical_plans"]), } AUDIO = { # ~500K "clean_speech": (0.60, [""]), # e.g. voxpopuli CC0 "noisy_conv": (0.20, [""]), "sound_events": (0.20, ["SYNTH:dsp_events"]), } THREED = { # ~400K, PAIRED multiview+poses <-> SLAT <-> text "primitives": (0.30, ["SYNTH:primitives"]), "props": (0.40, ["LOCAL:cc0_assets:prop", "SYNTH:props"]), "organic": (0.30, ["LOCAL:cc0_assets:char"]), } TARGET = {"vision": 800_000, "audio": 500_000, "3d": 400_000} def sha256_str(s): return hashlib.sha256(s.encode()).hexdigest()[:16] # ---------- TIER-A GATE ---------- def gate(license_str, source): lic = (license_str or "").strip().lower().replace(" ", "-") if lic in TIER_A: return True raise ValueError(f"TIER-A GATE REJECT: source={source} license={license_str!r} " f"(not in {sorted(TIER_A)}). Re-gate before use.") # ---------- caption / tag (pluggable; native-first) ---------- def caption_clean(item, modality): """Native caption if present+quality; else generate with a Tier-A captioner.""" if item.get("caption"): return item["caption"], "native" if modality in ("vision","3d"): from captioners import joycaption # JoyCaption Q4 GGUF (6900XT) — apache return joycaption(item["image"]), "joycaption" if modality == "audio": return item.get("transcript") or item.get("text") or "", "native_transcript" return "", "none" def tag_clean(item, modality): if item.get("tags"): return item["tags"] from taggers import siglip2_tags # prithiv SigLIP2 classifiers — offline labelers return siglip2_tags(item, modality) # ---------- 3D input-prep ---------- def render_multiview(mesh_path, n=4): from render3d import render_views # blender/kaolin: glb/mesh -> N views + cam poses return render_views(mesh_path, n) def encode_slat(mesh_path): """glb/mesh -> TRELLIS SLAT latents (voxel feats via DINOv2-on-voxels -> SLatEncoder).""" from slat_prep import mesh_to_slat_inputs # -> (coords[M,4], feats[M,1024]) import forward as F coords, feats = mesh_to_slat_inputs(mesh_path) gf, gc = F._MODEL.enc_geom(coords, feats) # frozen SLAT encoder (staged) xyz = (gc[:,[1,2,3]].float()/gc[:,[1,2,3]].float().max().clamp(min=1)*15).long() return gf, xyz # ---------- streaming loaders ---------- def stream_source(src, modality, quota): """Yield up to `quota` dict items from a source spec, each already TIER-A gated.""" if src.startswith(" gate its license, then stream from datasets import load_dataset from huggingface_hub import dataset_info gate(getattr(dataset_info(src), "card_data", {}).get("license"), src) ds = load_dataset(src, split="train", streaming=True) for i, row in zip(range(quota), ds): yield row # ---------- curate one modality ---------- def curate(modality, spec, target, model): os.makedirs(f"{CACHE}/{modality}", exist_ok=True) idx_path = f"{CACHE}/index.jsonl" manifest, n = [], 0 for bucket, (ratio, sources) in spec.items(): quota = int(target * ratio) per = max(1, quota // max(1, len(sources))) for src in sources: got = 0 for item in stream_source(src, modality, per): cap, cap_src = caption_clean(item, modality) if not cap: continue tags = tag_clean(item, modality) iid = f"{modality[0]}{n}_{sha256_str(src+cap)}" # encode with the frozen tower -> cache (mirrors preextract.py format) # (vision: enc_vision(px); audio: enc_audio(mel,lens); 3d: encode_slat(mesh)) # ... writes {CACHE}/{modality}/{iid}.pt and appends index row ... manifest.append({"id": iid, "modality": modality, "bucket": bucket, "source": src, "caption_src": cap_src, "caption": cap, "tags": tags, "sha": sha256_str(src+cap)}) n += 1; got += 1 print(f" [{modality}] {bucket} <- {src}: {got}") json.dump({"modality": modality, "target": target, "curated": n, "items": manifest}, open(f"{CACHE}/{modality}_manifest.json", "w"), indent=2) print(f"{modality}: curated {n}/{target}") return n if __name__ == "__main__": if "--dry-run" in sys.argv: # plan + gate self-test; no model load, no data pull print("=== SLIVER COMPOSITION (Tier-A gated) ===") for mod, spec in [("vision", VISION), ("audio", AUDIO), ("3d", THREED)]: print(f"{mod} (target {TARGET[mod]:,}):") for b,(r,srcs) in spec.items(): print(f" {b:16s} {int(r*100):3d}% {srcs}") print("=== TIER-A GATE self-test ===") for lic,exp in [("apache-2.0",True),("mit",True),("cc0-1.0",True),("cc-by-4.0",False), ("odc-by",False),("other",False),("cc-by-nc-4.0",False)]: try: gate(lic,"test"); got=True except ValueError: got=False print(f" {lic:14s} accept={got} {'OK' if got==exp else 'FAIL'}") print("DRY-RUN OK — code valid, gate correct. Stage a Tier-A corpus + drop --dry-run to curate.") sys.exit(0) import forward as F model = F.AetherPhase1().to(F.dev).eval(); F._MODEL = model total = 0 for mod, spec in [("vision", VISION), ("audio", AUDIO), ("3d", THREED)]: total += curate(mod, spec, TARGET[mod], model) print(f"SLIVER CURATION: {total} items -> {CACHE} (feed train_cached.py).") print("Sources marked need a Tier-A corpus staged; SYNTH/LOCAL = our-own (generate-own).")