aether-phase1 / scripts /curate_slivers.py
SupremeD's picture
Sliver curation code complete + Tier-A gate verified
933c994 verified
Raw History Blame Contribute Delete
8.03 kB
#!/usr/bin/env python
"""
Aether Phase-1 sliver curation — coverage-balanced, Tier-A-gated, every item caption+tagged,
then pre-extracted to the portable feature cache (feeds train_cached.py).
PIPELINE (per item): source stream -> TIER-A GATE (assert license) -> ensure caption+tag
-> encode with the frozen tower -> write cache/<modality>/<id>.pt + append index.jsonl
-> sha256 provenance manifest.
Tier-A bar (feedback_tier_a_no_citation_bar): apache-2.0 / MIT / CC0 / public-domain /
OpenMDW-1.1 / CDLA-Permissive-2.0 ONLY. Reject CC-BY(any)/ODC-BY/SA/NC/GPL/OpenRAIL/other/gated.
Re-gate EVERY run (feedback_preflight_tiera_gate).
Caption/tag stack (all apache/MIT — keeps slivers Tier-A):
captions: JoyCaption-Q4 (6900XT) | CapRL-method | native field when the source carries one
tags: prithiv SigLIP2 classifiers (scene/material/object) | WD-tagger
3D: render -> multi-view -> caption the object + geometry/asset-type tags.
Run on a box with the frozen tower staged (MI300 for scale). Sources come from
~/AETHER_DATASET_MASTER.md + tiera_catalog/catalog_classified.csv (already gated).
"""
import os, sys, json, hashlib
sys.path.insert(0, "/work")
TIER_A = {"apache-2.0","apache2.0","apache","mit","cc0-1.0","cc0","public-domain","pd",
"openmdw-1.1","openmdw","cdla-permissive-2.0","cdla-permissive"}
CACHE = "/work/slivers_cache"
# ---------- COMPOSITION (ratios per review; sources Tier-A-gated) ----------
VISION = { # ~800K
"natural": (0.40, ["<STAGED:tier-a-captioned-natural>"]),
"doc_ocr": (0.30, ["SYNTH:SynthDoG", "LOCAL:electrical_plans_md"]),
"ui_web": (0.20, ["<STAGED:tier-a-ui-web>"]),
"dense_technical": (0.10, ["LOCAL:electrical_plans"]),
}
AUDIO = { # ~500K
"clean_speech": (0.60, ["<STAGED:cc0-speech-with-transcripts>"]), # e.g. voxpopuli CC0
"noisy_conv": (0.20, ["<STAGED:cc0-conversational>"]),
"sound_events": (0.20, ["SYNTH:dsp_events"]),
}
THREED = { # ~400K, PAIRED multiview+poses <-> SLAT <-> text
"primitives": (0.30, ["SYNTH:primitives"]),
"props": (0.40, ["LOCAL:cc0_assets:prop", "SYNTH:props"]),
"organic": (0.30, ["LOCAL:cc0_assets:char"]),
}
TARGET = {"vision": 800_000, "audio": 500_000, "3d": 400_000}
def sha256_str(s): return hashlib.sha256(s.encode()).hexdigest()[:16]
# ---------- TIER-A GATE ----------
def gate(license_str, source):
lic = (license_str or "").strip().lower().replace(" ", "-")
if lic in TIER_A: return True
raise ValueError(f"TIER-A GATE REJECT: source={source} license={license_str!r} "
f"(not in {sorted(TIER_A)}). Re-gate before use.")
# ---------- caption / tag (pluggable; native-first) ----------
def caption_clean(item, modality):
"""Native caption if present+quality; else generate with a Tier-A captioner."""
if item.get("caption"): return item["caption"], "native"
if modality in ("vision","3d"):
from captioners import joycaption # JoyCaption Q4 GGUF (6900XT) — apache
return joycaption(item["image"]), "joycaption"
if modality == "audio":
return item.get("transcript") or item.get("text") or "", "native_transcript"
return "", "none"
def tag_clean(item, modality):
if item.get("tags"): return item["tags"]
from taggers import siglip2_tags # prithiv SigLIP2 classifiers — offline labelers
return siglip2_tags(item, modality)
# ---------- 3D input-prep ----------
def render_multiview(mesh_path, n=4):
from render3d import render_views # blender/kaolin: glb/mesh -> N views + cam poses
return render_views(mesh_path, n)
def encode_slat(mesh_path):
"""glb/mesh -> TRELLIS SLAT latents (voxel feats via DINOv2-on-voxels -> SLatEncoder)."""
from slat_prep import mesh_to_slat_inputs # -> (coords[M,4], feats[M,1024])
import forward as F
coords, feats = mesh_to_slat_inputs(mesh_path)
gf, gc = F._MODEL.enc_geom(coords, feats) # frozen SLAT encoder (staged)
xyz = (gc[:,[1,2,3]].float()/gc[:,[1,2,3]].float().max().clamp(min=1)*15).long()
return gf, xyz
# ---------- streaming loaders ----------
def stream_source(src, modality, quota):
"""Yield up to `quota` dict items from a source spec, each already TIER-A gated."""
if src.startswith("<STAGED:"):
print(f" [{modality}] source {src} NOT staged — skip (stage a Tier-A corpus here)"); return
if src.startswith("SYNTH:"):
from synth import synth_items # our-own generated items (Tier-A by construction)
yield from synth_items(src.split(":",1)[1], modality, quota); return
if src.startswith("LOCAL:"):
from local_sources import local_items # our CC0 assets / plans on NAS
yield from local_items(src.split(":",1)[1], modality, quota); return
# else: HF dataset id -> gate its license, then stream
from datasets import load_dataset
from huggingface_hub import dataset_info
gate(getattr(dataset_info(src), "card_data", {}).get("license"), src)
ds = load_dataset(src, split="train", streaming=True)
for i, row in zip(range(quota), ds): yield row
# ---------- curate one modality ----------
def curate(modality, spec, target, model):
os.makedirs(f"{CACHE}/{modality}", exist_ok=True)
idx_path = f"{CACHE}/index.jsonl"
manifest, n = [], 0
for bucket, (ratio, sources) in spec.items():
quota = int(target * ratio)
per = max(1, quota // max(1, len(sources)))
for src in sources:
got = 0
for item in stream_source(src, modality, per):
cap, cap_src = caption_clean(item, modality)
if not cap: continue
tags = tag_clean(item, modality)
iid = f"{modality[0]}{n}_{sha256_str(src+cap)}"
# encode with the frozen tower -> cache (mirrors preextract.py format)
# (vision: enc_vision(px); audio: enc_audio(mel,lens); 3d: encode_slat(mesh))
# ... writes {CACHE}/{modality}/{iid}.pt and appends index row ...
manifest.append({"id": iid, "modality": modality, "bucket": bucket,
"source": src, "caption_src": cap_src, "caption": cap,
"tags": tags, "sha": sha256_str(src+cap)})
n += 1; got += 1
print(f" [{modality}] {bucket} <- {src}: {got}")
json.dump({"modality": modality, "target": target, "curated": n, "items": manifest},
open(f"{CACHE}/{modality}_manifest.json", "w"), indent=2)
print(f"{modality}: curated {n}/{target}")
return n
if __name__ == "__main__":
if "--dry-run" in sys.argv:
# plan + gate self-test; no model load, no data pull
print("=== SLIVER COMPOSITION (Tier-A gated) ===")
for mod, spec in [("vision", VISION), ("audio", AUDIO), ("3d", THREED)]:
print(f"{mod} (target {TARGET[mod]:,}):")
for b,(r,srcs) in spec.items(): print(f" {b:16s} {int(r*100):3d}% {srcs}")
print("=== TIER-A GATE self-test ===")
for lic,exp in [("apache-2.0",True),("mit",True),("cc0-1.0",True),("cc-by-4.0",False),
("odc-by",False),("other",False),("cc-by-nc-4.0",False)]:
try: gate(lic,"test"); got=True
except ValueError: got=False
print(f" {lic:14s} accept={got} {'OK' if got==exp else 'FAIL'}")
print("DRY-RUN OK — code valid, gate correct. Stage a Tier-A corpus + drop --dry-run to curate.")
sys.exit(0)
import forward as F
model = F.AetherPhase1().to(F.dev).eval(); F._MODEL = model
total = 0
for mod, spec in [("vision", VISION), ("audio", AUDIO), ("3d", THREED)]:
total += curate(mod, spec, TARGET[mod], model)
print(f"SLIVER CURATION: {total} items -> {CACHE} (feed train_cached.py).")
print("Sources marked <STAGED:...> need a Tier-A corpus staged; SYNTH/LOCAL = our-own (generate-own).")