#!/usr/bin/env python """Batch Cupid3D MULTI-VIEW driver (paper Fig. 7 / Sec. 6 test-time extension). Paper (arXiv 2510.20776 v2, Fig. 7 caption): "When multiple input views are available, we fuse the shared view-agnostic object latent across flow paths (similar to MultiDiffusion [2]), enabling object and cameras refinement across all views." Sec. 6: "From multiple images, our method refines 3D reconstructions to align with all observations by fusing a shared object latent during sampling, similar to Multi-Diffusion [2]." The released code (github cupid3d/Cupid @10af9b2, only commit/branch upstream) exposes only single-image `Cupid3DPipeline.run()`; there is NO multi-view entry point. This driver implements the paper's procedure with the released pipeline pieces, unchanged weights and sampler settings (pipeline.json: 25 Euler steps, rescale_t 3, CFG 5 on t in [0.5, 1]), and MultiDiffusion fusion = per-step average of the per-view updates on the shared variables: Stage 1 (sparse-structure flow, z_s = [occupancy latent (view-agnostic) | UV/2D-3D-correspondence latent (view-specific)], split exactly as SparseStructureDecoder.decode splits it): one flow path per view (own DINOv2 cond, own UV-noise); the occupancy channels share one noise and after every Euler step are replaced by the mean over views. The UV channels stay per-view -> per-view camera pose via the released decode_uv (DLT), all expressed in ONE shared canonical object frame (the "SfM-like" output of Fig. 7). Stage 2 (SLAT flow on the fused coords, canonical/view-agnostic): one shared noise; each view path gets its own pose-aligned conditioning (UVs projected with that view's pose, its own visual_cond + dino_cond); every Euler step uses the mean of the per-view (CFG-guided) velocities. Views are run as batch-1 forwards in a loop (memory == the 1v pipeline). With --views 1 this reduces exactly to Cupid3DPipeline.run() semantics. Usage: python batch_cupid_mv.py --selection SEL.json --inputs DIR --out OUTDIR --views 4 [--seed 42] [--shard i --nshards n] [--limit N] (--exp EXPDIR = shorthand for --selection EXPDIR/selection.json --inputs EXPDIR/inputs) Inputs: DIR/_{front,side,back,oside}.png RGBA crops, alpha used as-is. Output: OUTDIR/.glb (Cupid canonical frame; to_glb simplify 0.95, tex 1024, identical to batch_cupid.py / Cupid save_mesh()) OUTDIR/.pose.json per-view Cupid camera (extrinsic 4x4 OpenCV world(canonical)->camera, intrinsic 3x3 in NORMALIZED image coords of the pad_to_square'd input image), plus pad/crop bookkeeping. Idempotent (skips existing .glb), per-object try/except, re-execs in cupid env. """ import os import sys ENV = "/lp-dev/jonghoon/mv-mesh/baselines/cupid/env" ENV_PY = os.path.join(ENV, "bin", "python") REPO = "/lp-dev/jonghoon/mv-mesh/baselines/cupid/repo" HF = "/lp-dev/jonghoon/mv-mesh/hf_cache" _ENV_VARS = { "PYTHONUNBUFFERED": "1", "OMP_NUM_THREADS": "4", "MKL_NUM_THREADS": "4", "SPCONV_ALGO": "native", "ATTN_BACKEND": "flash_attn", "MKL_THREADING_LAYER": "GNU", "MKL_SERVICE_FORCE_INTEL": "0", "HF_HOME": HF, "HUGGINGFACE_HUB_CACHE": HF, "HF_HUB_CACHE": HF, "TORCH_HOME": "/lp-dev/jonghoon/mv-mesh/torch_hub", "CUDA_HOME": "/usr/local/cuda-12.8", } def _reexec_in_env(): env = dict(os.environ) for k, v in _ENV_VARS.items(): env[k] = v for k in ("OMP_NUM_THREADS", "MKL_NUM_THREADS"): if os.environ.get(k): env[k] = os.environ[k] env["CONDA_PREFIX"] = ENV env["PATH"] = os.path.join(ENV, "bin") + os.pathsep + "/usr/local/cuda-12.8/bin" + os.pathsep + env.get("PATH", "") if not env.get("CUDA_VISIBLE_DEVICES"): sys.exit("ERROR: set CUDA_VISIBLE_DEVICES explicitly (shared box).") env["_CUPID_BATCH_INENV"] = "1" print(f"[mv] re-exec in {ENV_PY} (CUDA_VISIBLE_DEVICES={env.get('CUDA_VISIBLE_DEVICES')})", flush=True) os.execve(ENV_PY, [ENV_PY, os.path.abspath(__file__)] + sys.argv[1:], env) if not os.environ.get("_CUPID_BATCH_INENV"): if not os.path.isfile(ENV_PY): sys.exit(f"ERROR: env python not found: {ENV_PY}") _reexec_in_env() for _k, _v in _ENV_VARS.items(): os.environ.setdefault(_k, _v) import argparse import json import time import traceback import numpy as np os.chdir(REPO) sys.path.insert(0, REPO) VIEW_TAGS = {1: ["front"], 2: ["front", "side"], 4: ["front", "side", "back", "oside"]} def _euler_tseq(steps, rescale_t): t_seq = np.linspace(1, 0, steps + 1) t_seq = rescale_t * t_seq / (1 + (rescale_t - 1) * t_seq) return [(float(t_seq[i]), float(t_seq[i + 1])) for i in range(steps)] def run_multiview(pipeline, images, seed=42): import torch with torch.no_grad(): return _run_multiview(pipeline, images, seed) def _run_multiview(pipeline, images, seed=42): """Cupid multi-view inference (MultiDiffusion-style fusion of the shared object latent across per-view flow paths). images: list of PIL RGBA (padded to square, as sample_utils.load_image returns). Returns dict with 'mesh','gaussian' (1 fused object) and per-view 'pose' + crop params.""" import torch from cupid.modules import sparse as sp dev = pipeline.device V = len(images) torch.manual_seed(seed) # ---- per-view preprocessing + conditioning (as in run()) ---- procs, conds, visuals = [], [], [] for im in images: p = pipeline.crop_image(pipeline.preprocess_image(im)) c, vis = pipeline.get_cond([p]) # {'cond','neg_cond'}, [1,3,256,256] procs.append(p); conds.append(c); visuals.append(vis) # ---- Stage 1: sparse structure + UV (pose) flow, fused occupancy ---- fm = pipeline.models['sparse_structure_flow_model'] res, C = fm.resolution, fm.in_channels n_ss = pipeline.structure_decoder.structure_decoder.latent_channels sampler = pipeline.sparse_structure_sampler params = dict(pipeline.sparse_structure_sampler_params) noise = torch.randn(V, C, res, res, res).to(dev) # CPU RNG, as run() noise[:, :n_ss] = noise[:1, :n_ss] # one shared object (occupancy) latent x = [noise[v:v + 1].clone() for v in range(V)] for t, t_prev in _euler_tseq(params['steps'], params['rescale_t']): for v in range(V): pv = sampler._inference_model(fm, x[v], t, cond=conds[v]['cond'], neg_cond=conds[v]['neg_cond'], cfg_strength=params['cfg_strength'], cfg_interval=params['cfg_interval']) x[v] = x[v] - (t - t_prev) * pv shared = torch.stack([xv[:, :n_ss] for xv in x]).mean(0) # MultiDiffusion fuse for v in range(V): x[v][:, :n_ss] = shared z_s = torch.cat(x, 0) st = pipeline.decode_zs(z_s) # coords identical across views poses = pipeline.decode_uv(st['uvs']) # one CameraPose per view (DLT) coords0 = st['coords'][st['coords'][:, 0] == 0].clone() # ---- Stage 2: SLAT flow, one shared latent, per-view pose-aligned cond ---- sm = pipeline.models['slat_flow_model'] ssampler = pipeline.slat_sampler sparams = dict(pipeline.slat_sampler_params) feats = torch.randn(coords0.shape[0], sm.out_channels).to(dev) xs = sp.SparseTensor(feats=feats, coords=coords0) view_cond = [] for v in range(V): uvs = pipeline._make_sparse_uvs_from_pose( xs, poses[v].extrinsic[None].to(dev), poses[v].intrinsic[None].to(dev)) mc = {'visual_cond': visuals[v], 'dino_cond': conds[v]['cond']} nc = {'visual_cond': torch.zeros_like(visuals[v]), 'dino_cond': torch.zeros_like(conds[v]['cond'])} view_cond.append((uvs, mc, nc)) for t, t_prev in _euler_tseq(sparams['steps'], sparams['rescale_t']): vsum = None for uvs, mc, nc in view_cond: pv = ssampler._inference_model(sm, xs, t, cond=mc, neg_cond=nc, cfg_strength=sparams['cfg_strength'], cfg_interval=sparams['cfg_interval'], uvs=uvs) vsum = pv.feats if vsum is None else vsum + pv.feats xs = xs.replace(xs.feats - (t - t_prev) * (vsum / V)) # fused update std = torch.tensor(pipeline.slat_normalization['std'])[None].to(dev) mean = torch.tensor(pipeline.slat_normalization['mean'])[None].to(dev) slat = xs * std + mean out = pipeline.decode_slat(slat, ['mesh', 'gaussian']) out['pose'] = [po.de_crop(pr.crop_params).as_dict() for po, pr in zip(poses, procs)] out['crop_params'] = [pr.crop_params.as_tuple() for pr in procs] out['n_coords'] = int(coords0.shape[0]) return out def main(): ap = argparse.ArgumentParser() ap.add_argument("--selection") ap.add_argument("--inputs") ap.add_argument("--exp", help="EXPDIR with selection.json + inputs/") ap.add_argument("--out", required=True) ap.add_argument("--views", type=int, default=4, choices=sorted(VIEW_TAGS)) ap.add_argument("--seed", type=int, default=42) ap.add_argument("--limit", type=int, default=None) ap.add_argument("--simplify", type=float, default=0.95) ap.add_argument("--texture-size", type=int, default=1024) ap.add_argument("--shard", type=int, default=0) ap.add_argument("--nshards", type=int, default=1) ap.add_argument("--reverse", action="store_true") args = ap.parse_args() if args.exp: args.selection = args.selection or os.path.join(args.exp, "selection.json") args.inputs = args.inputs or os.path.join(args.exp, "inputs") if not (args.selection and args.inputs): ap.error("need --exp or --selection + --inputs") with open(args.selection) as f: sel = json.load(f) sel = sel["selections"] if isinstance(sel, dict) else sel objects = [s["object"] for s in sel] if args.limit: objects = objects[: args.limit] if args.nshards > 1: objects = objects[args.shard::args.nshards] if args.reverse: objects = objects[::-1] tags = VIEW_TAGS[args.views] outdir = os.path.abspath(args.out) os.makedirs(outdir, exist_ok=True) todo = [o for o in objects if not os.path.isfile(os.path.join(outdir, f"{o}.glb"))] print(f"[mv] {len(objects)} object(s), {len(objects)-len(todo)} done, {len(todo)} to run " f"| views={args.views} tags={tags} seed={args.seed}", flush=True) import torch from PIL import Image from cupid.pipelines import Cupid3DPipeline from cupid.utils import sample_utils, postprocessing_utils n_ok = n_fail = n_skip = 0 pipeline = None for i, name in enumerate(objects, 1): out = os.path.join(outdir, f"{name}.glb") t0 = time.time() try: if os.path.isfile(out): n_skip += 1 print(f"[{i}/{len(objects)}] {name} SKIP", flush=True) continue paths = [os.path.join(args.inputs, f"{name}_{t}.png") for t in tags] for p in paths: if not os.path.isfile(p): raise FileNotFoundError(p) if pipeline is None: tl = time.time() pipeline = Cupid3DPipeline.from_pretrained("hbb1/Cupid") pipeline.cuda() print(f"[mv] pipeline ready in {time.time()-tl:.1f}s", flush=True) t0 = time.time() torch.cuda.reset_peak_memory_stats() raw_sizes = [Image.open(p).size for p in paths] images = [sample_utils.load_image(p) for p in paths] # pad_to_square res = run_multiview(pipeline, images, seed=args.seed) glb = postprocessing_utils.to_glb(res["gaussian"][0], res["mesh"][0], simplify=args.simplify, texture_size=args.texture_size, verbose=False) views = [] for t, p, (W, H), im, pose, cp in zip(tags, paths, raw_sizes, images, res["pose"], res["crop_params"]): S = im.size[0] views.append({ "tag": t, "image": p, "raw_size_wh": [W, H], "padded_size": S, "pad_offset_xy": [(S - W) // 2, (S - H) // 2], "cupid_crop_params": {"fov_scale": cp[0], "cx_offset": cp[1], "cy_offset": cp[2]}, "extrinsic": pose["extrinsic"].detach().cpu().numpy().tolist(), "intrinsic_normalized": pose["intrinsic"].detach().cpu().numpy().tolist(), }) pose_json = { "object": name, "views": args.views, "seed": args.seed, "method": "cupid multi-view (MultiDiffusion fusion of shared object latent; paper Fig.7)", "frame_note": ("extrinsic = OpenCV world->camera, world = Cupid canonical voxel frame " "(xyz in [-0.5,0.5]^3, the frame of Cupid's decoded mesh BEFORE to_glb's " "Y-up conversion); intrinsic maps to normalized [0,1] uv of the " "pad_to_square'd input image (multiply row0/1 by padded_size for pixels)."), "n_coords": res["n_coords"], "per_view": views, } tmp = out + ".tmp.glb" glb.export(tmp) with open(out[:-4] + ".pose.json", "w") as f: json.dump(pose_json, f, indent=1) os.replace(tmp, out) peak = torch.cuda.max_memory_allocated() / 2**30 del res, glb, images n_ok += 1 print(f"[{i}/{len(objects)}] {name} OK {time.time()-t0:.1f}s peak_alloc={peak:.2f}GiB -> {out}", flush=True) except Exception: n_fail += 1 traceback.print_exc() print(f"[{i}/{len(objects)}] {name} FAIL {time.time()-t0:.1f}s", flush=True) finally: try: torch.cuda.empty_cache() except Exception: pass print(f"[mv] DONE ok={n_ok} fail={n_fail} skip={n_skip}", flush=True) if __name__ == "__main__": main()