bert_simpson / forgebench /code /baselines /batch_cupid_mv.py
Ronaldo-GOAT's picture
forgebench: ours batched pipeline + equivalence report, Cupid multi-view driver
84bff06 verified
Raw History Blame Contribute Delete
14.4 kB
#!/usr/bin/env python
"""Batch Cupid3D MULTI-VIEW driver (paper Fig. 7 / Sec. 6 test-time extension).
Paper (arXiv 2510.20776 v2, Fig. 7 caption): "When multiple input views are
available, we fuse the shared view-agnostic object latent across flow paths
(similar to MultiDiffusion [2]), enabling object and cameras refinement across
all views." Sec. 6: "From multiple images, our method refines 3D
reconstructions to align with all observations by fusing a shared object latent
during sampling, similar to Multi-Diffusion [2]."
The released code (github cupid3d/Cupid @10af9b2, only commit/branch upstream)
exposes only single-image `Cupid3DPipeline.run()`; there is NO multi-view entry
point. This driver implements the paper's procedure with the released pipeline
pieces, unchanged weights and sampler settings (pipeline.json: 25 Euler steps,
rescale_t 3, CFG 5 on t in [0.5, 1]), and MultiDiffusion fusion = per-step
average of the per-view updates on the shared variables:
Stage 1 (sparse-structure flow, z_s = [occupancy latent (view-agnostic) |
UV/2D-3D-correspondence latent (view-specific)], split exactly as
SparseStructureDecoder.decode splits it):
one flow path per view (own DINOv2 cond, own UV-noise); the occupancy
channels share one noise and after every Euler step are replaced by the
mean over views. The UV channels stay per-view -> per-view camera pose via
the released decode_uv (DLT), all expressed in ONE shared canonical object
frame (the "SfM-like" output of Fig. 7).
Stage 2 (SLAT flow on the fused coords, canonical/view-agnostic): one shared
noise; each view path gets its own pose-aligned conditioning (UVs projected
with that view's pose, its own visual_cond + dino_cond); every Euler step
uses the mean of the per-view (CFG-guided) velocities.
Views are run as batch-1 forwards in a loop (memory == the 1v pipeline).
With --views 1 this reduces exactly to Cupid3DPipeline.run() semantics.
Usage:
python batch_cupid_mv.py --selection SEL.json --inputs DIR --out OUTDIR --views 4
[--seed 42] [--shard i --nshards n] [--limit N]
(--exp EXPDIR = shorthand for --selection EXPDIR/selection.json --inputs EXPDIR/inputs)
Inputs: DIR/<object>_{front,side,back,oside}.png RGBA crops, alpha used as-is.
Output: OUTDIR/<object>.glb (Cupid canonical frame; to_glb simplify 0.95, tex 1024,
identical to batch_cupid.py / Cupid save_mesh())
OUTDIR/<object>.pose.json per-view Cupid camera (extrinsic 4x4 OpenCV
world(canonical)->camera, intrinsic 3x3 in NORMALIZED image coords of the
pad_to_square'd input image), plus pad/crop bookkeeping.
Idempotent (skips existing .glb), per-object try/except, re-execs in cupid env.
"""
import os
import sys
ENV = "/lp-dev/jonghoon/mv-mesh/baselines/cupid/env"
ENV_PY = os.path.join(ENV, "bin", "python")
REPO = "/lp-dev/jonghoon/mv-mesh/baselines/cupid/repo"
HF = "/lp-dev/jonghoon/mv-mesh/hf_cache"
_ENV_VARS = {
"PYTHONUNBUFFERED": "1",
"OMP_NUM_THREADS": "4",
"MKL_NUM_THREADS": "4",
"SPCONV_ALGO": "native",
"ATTN_BACKEND": "flash_attn",
"MKL_THREADING_LAYER": "GNU",
"MKL_SERVICE_FORCE_INTEL": "0",
"HF_HOME": HF,
"HUGGINGFACE_HUB_CACHE": HF,
"HF_HUB_CACHE": HF,
"TORCH_HOME": "/lp-dev/jonghoon/mv-mesh/torch_hub",
"CUDA_HOME": "/usr/local/cuda-12.8",
}
def _reexec_in_env():
env = dict(os.environ)
for k, v in _ENV_VARS.items():
env[k] = v
for k in ("OMP_NUM_THREADS", "MKL_NUM_THREADS"):
if os.environ.get(k):
env[k] = os.environ[k]
env["CONDA_PREFIX"] = ENV
env["PATH"] = os.path.join(ENV, "bin") + os.pathsep + "/usr/local/cuda-12.8/bin" + os.pathsep + env.get("PATH", "")
if not env.get("CUDA_VISIBLE_DEVICES"):
sys.exit("ERROR: set CUDA_VISIBLE_DEVICES explicitly (shared box).")
env["_CUPID_BATCH_INENV"] = "1"
print(f"[mv] re-exec in {ENV_PY} (CUDA_VISIBLE_DEVICES={env.get('CUDA_VISIBLE_DEVICES')})", flush=True)
os.execve(ENV_PY, [ENV_PY, os.path.abspath(__file__)] + sys.argv[1:], env)
if not os.environ.get("_CUPID_BATCH_INENV"):
if not os.path.isfile(ENV_PY):
sys.exit(f"ERROR: env python not found: {ENV_PY}")
_reexec_in_env()
for _k, _v in _ENV_VARS.items():
os.environ.setdefault(_k, _v)
import argparse
import json
import time
import traceback
import numpy as np
os.chdir(REPO)
sys.path.insert(0, REPO)
VIEW_TAGS = {1: ["front"], 2: ["front", "side"],
4: ["front", "side", "back", "oside"]}
def _euler_tseq(steps, rescale_t):
t_seq = np.linspace(1, 0, steps + 1)
t_seq = rescale_t * t_seq / (1 + (rescale_t - 1) * t_seq)
return [(float(t_seq[i]), float(t_seq[i + 1])) for i in range(steps)]
def run_multiview(pipeline, images, seed=42):
import torch
with torch.no_grad():
return _run_multiview(pipeline, images, seed)
def _run_multiview(pipeline, images, seed=42):
"""Cupid multi-view inference (MultiDiffusion-style fusion of the shared
object latent across per-view flow paths). images: list of PIL RGBA (padded
to square, as sample_utils.load_image returns). Returns dict with
'mesh','gaussian' (1 fused object) and per-view 'pose' + crop params."""
import torch
from cupid.modules import sparse as sp
dev = pipeline.device
V = len(images)
torch.manual_seed(seed)
# ---- per-view preprocessing + conditioning (as in run()) ----
procs, conds, visuals = [], [], []
for im in images:
p = pipeline.crop_image(pipeline.preprocess_image(im))
c, vis = pipeline.get_cond([p]) # {'cond','neg_cond'}, [1,3,256,256]
procs.append(p); conds.append(c); visuals.append(vis)
# ---- Stage 1: sparse structure + UV (pose) flow, fused occupancy ----
fm = pipeline.models['sparse_structure_flow_model']
res, C = fm.resolution, fm.in_channels
n_ss = pipeline.structure_decoder.structure_decoder.latent_channels
sampler = pipeline.sparse_structure_sampler
params = dict(pipeline.sparse_structure_sampler_params)
noise = torch.randn(V, C, res, res, res).to(dev) # CPU RNG, as run()
noise[:, :n_ss] = noise[:1, :n_ss] # one shared object (occupancy) latent
x = [noise[v:v + 1].clone() for v in range(V)]
for t, t_prev in _euler_tseq(params['steps'], params['rescale_t']):
for v in range(V):
pv = sampler._inference_model(fm, x[v], t, cond=conds[v]['cond'],
neg_cond=conds[v]['neg_cond'],
cfg_strength=params['cfg_strength'],
cfg_interval=params['cfg_interval'])
x[v] = x[v] - (t - t_prev) * pv
shared = torch.stack([xv[:, :n_ss] for xv in x]).mean(0) # MultiDiffusion fuse
for v in range(V):
x[v][:, :n_ss] = shared
z_s = torch.cat(x, 0)
st = pipeline.decode_zs(z_s) # coords identical across views
poses = pipeline.decode_uv(st['uvs']) # one CameraPose per view (DLT)
coords0 = st['coords'][st['coords'][:, 0] == 0].clone()
# ---- Stage 2: SLAT flow, one shared latent, per-view pose-aligned cond ----
sm = pipeline.models['slat_flow_model']
ssampler = pipeline.slat_sampler
sparams = dict(pipeline.slat_sampler_params)
feats = torch.randn(coords0.shape[0], sm.out_channels).to(dev)
xs = sp.SparseTensor(feats=feats, coords=coords0)
view_cond = []
for v in range(V):
uvs = pipeline._make_sparse_uvs_from_pose(
xs, poses[v].extrinsic[None].to(dev), poses[v].intrinsic[None].to(dev))
mc = {'visual_cond': visuals[v], 'dino_cond': conds[v]['cond']}
nc = {'visual_cond': torch.zeros_like(visuals[v]),
'dino_cond': torch.zeros_like(conds[v]['cond'])}
view_cond.append((uvs, mc, nc))
for t, t_prev in _euler_tseq(sparams['steps'], sparams['rescale_t']):
vsum = None
for uvs, mc, nc in view_cond:
pv = ssampler._inference_model(sm, xs, t, cond=mc, neg_cond=nc,
cfg_strength=sparams['cfg_strength'],
cfg_interval=sparams['cfg_interval'],
uvs=uvs)
vsum = pv.feats if vsum is None else vsum + pv.feats
xs = xs.replace(xs.feats - (t - t_prev) * (vsum / V)) # fused update
std = torch.tensor(pipeline.slat_normalization['std'])[None].to(dev)
mean = torch.tensor(pipeline.slat_normalization['mean'])[None].to(dev)
slat = xs * std + mean
out = pipeline.decode_slat(slat, ['mesh', 'gaussian'])
out['pose'] = [po.de_crop(pr.crop_params).as_dict() for po, pr in zip(poses, procs)]
out['crop_params'] = [pr.crop_params.as_tuple() for pr in procs]
out['n_coords'] = int(coords0.shape[0])
return out
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--selection")
ap.add_argument("--inputs")
ap.add_argument("--exp", help="EXPDIR with selection.json + inputs/")
ap.add_argument("--out", required=True)
ap.add_argument("--views", type=int, default=4, choices=sorted(VIEW_TAGS))
ap.add_argument("--seed", type=int, default=42)
ap.add_argument("--limit", type=int, default=None)
ap.add_argument("--simplify", type=float, default=0.95)
ap.add_argument("--texture-size", type=int, default=1024)
ap.add_argument("--shard", type=int, default=0)
ap.add_argument("--nshards", type=int, default=1)
ap.add_argument("--reverse", action="store_true")
args = ap.parse_args()
if args.exp:
args.selection = args.selection or os.path.join(args.exp, "selection.json")
args.inputs = args.inputs or os.path.join(args.exp, "inputs")
if not (args.selection and args.inputs):
ap.error("need --exp or --selection + --inputs")
with open(args.selection) as f:
sel = json.load(f)
sel = sel["selections"] if isinstance(sel, dict) else sel
objects = [s["object"] for s in sel]
if args.limit:
objects = objects[: args.limit]
if args.nshards > 1:
objects = objects[args.shard::args.nshards]
if args.reverse:
objects = objects[::-1]
tags = VIEW_TAGS[args.views]
outdir = os.path.abspath(args.out)
os.makedirs(outdir, exist_ok=True)
todo = [o for o in objects if not os.path.isfile(os.path.join(outdir, f"{o}.glb"))]
print(f"[mv] {len(objects)} object(s), {len(objects)-len(todo)} done, {len(todo)} to run "
f"| views={args.views} tags={tags} seed={args.seed}", flush=True)
import torch
from PIL import Image
from cupid.pipelines import Cupid3DPipeline
from cupid.utils import sample_utils, postprocessing_utils
n_ok = n_fail = n_skip = 0
pipeline = None
for i, name in enumerate(objects, 1):
out = os.path.join(outdir, f"{name}.glb")
t0 = time.time()
try:
if os.path.isfile(out):
n_skip += 1
print(f"[{i}/{len(objects)}] {name} SKIP", flush=True)
continue
paths = [os.path.join(args.inputs, f"{name}_{t}.png") for t in tags]
for p in paths:
if not os.path.isfile(p):
raise FileNotFoundError(p)
if pipeline is None:
tl = time.time()
pipeline = Cupid3DPipeline.from_pretrained("hbb1/Cupid")
pipeline.cuda()
print(f"[mv] pipeline ready in {time.time()-tl:.1f}s", flush=True)
t0 = time.time()
torch.cuda.reset_peak_memory_stats()
raw_sizes = [Image.open(p).size for p in paths]
images = [sample_utils.load_image(p) for p in paths] # pad_to_square
res = run_multiview(pipeline, images, seed=args.seed)
glb = postprocessing_utils.to_glb(res["gaussian"][0], res["mesh"][0],
simplify=args.simplify,
texture_size=args.texture_size, verbose=False)
views = []
for t, p, (W, H), im, pose, cp in zip(tags, paths, raw_sizes, images,
res["pose"], res["crop_params"]):
S = im.size[0]
views.append({
"tag": t, "image": p, "raw_size_wh": [W, H], "padded_size": S,
"pad_offset_xy": [(S - W) // 2, (S - H) // 2],
"cupid_crop_params": {"fov_scale": cp[0], "cx_offset": cp[1], "cy_offset": cp[2]},
"extrinsic": pose["extrinsic"].detach().cpu().numpy().tolist(),
"intrinsic_normalized": pose["intrinsic"].detach().cpu().numpy().tolist(),
})
pose_json = {
"object": name, "views": args.views, "seed": args.seed,
"method": "cupid multi-view (MultiDiffusion fusion of shared object latent; paper Fig.7)",
"frame_note": ("extrinsic = OpenCV world->camera, world = Cupid canonical voxel frame "
"(xyz in [-0.5,0.5]^3, the frame of Cupid's decoded mesh BEFORE to_glb's "
"Y-up conversion); intrinsic maps to normalized [0,1] uv of the "
"pad_to_square'd input image (multiply row0/1 by padded_size for pixels)."),
"n_coords": res["n_coords"],
"per_view": views,
}
tmp = out + ".tmp.glb"
glb.export(tmp)
with open(out[:-4] + ".pose.json", "w") as f:
json.dump(pose_json, f, indent=1)
os.replace(tmp, out)
peak = torch.cuda.max_memory_allocated() / 2**30
del res, glb, images
n_ok += 1
print(f"[{i}/{len(objects)}] {name} OK {time.time()-t0:.1f}s peak_alloc={peak:.2f}GiB -> {out}", flush=True)
except Exception:
n_fail += 1
traceback.print_exc()
print(f"[{i}/{len(objects)}] {name} FAIL {time.time()-t0:.1f}s", flush=True)
finally:
try:
torch.cuda.empty_cache()
except Exception:
pass
print(f"[mv] DONE ok={n_ok} fail={n_fail} skip={n_skip}", flush=True)
if __name__ == "__main__":
main()