Download forgebench/code/baselines/batch_cupid_mv.py from Ronaldo-GOAT/bert_simpson: direct link, hf CLI and curl.
- Browser
- Download file 14.4 kB
-
https://huggingface.co/Ronaldo-GOAT/bert_simpson/resolve/main/forgebench/code/baselines/batch_cupid_mv.py
- Command line
-
hf download hf://Ronaldo-GOAT/bert_simpson/forgebench/code/baselines/batch_cupid_mv.py
-
curl -L -o batch_cupid_mv.py https://huggingface.co/Ronaldo-GOAT/bert_simpson/resolve/main/forgebench/code/baselines/batch_cupid_mv.py
14.4 kB
| #!/usr/bin/env python | |
| """Batch Cupid3D MULTI-VIEW driver (paper Fig. 7 / Sec. 6 test-time extension). | |
| Paper (arXiv 2510.20776 v2, Fig. 7 caption): "When multiple input views are | |
| available, we fuse the shared view-agnostic object latent across flow paths | |
| (similar to MultiDiffusion [2]), enabling object and cameras refinement across | |
| all views." Sec. 6: "From multiple images, our method refines 3D | |
| reconstructions to align with all observations by fusing a shared object latent | |
| during sampling, similar to Multi-Diffusion [2]." | |
| The released code (github cupid3d/Cupid @10af9b2, only commit/branch upstream) | |
| exposes only single-image `Cupid3DPipeline.run()`; there is NO multi-view entry | |
| point. This driver implements the paper's procedure with the released pipeline | |
| pieces, unchanged weights and sampler settings (pipeline.json: 25 Euler steps, | |
| rescale_t 3, CFG 5 on t in [0.5, 1]), and MultiDiffusion fusion = per-step | |
| average of the per-view updates on the shared variables: | |
| Stage 1 (sparse-structure flow, z_s = [occupancy latent (view-agnostic) | | |
| UV/2D-3D-correspondence latent (view-specific)], split exactly as | |
| SparseStructureDecoder.decode splits it): | |
| one flow path per view (own DINOv2 cond, own UV-noise); the occupancy | |
| channels share one noise and after every Euler step are replaced by the | |
| mean over views. The UV channels stay per-view -> per-view camera pose via | |
| the released decode_uv (DLT), all expressed in ONE shared canonical object | |
| frame (the "SfM-like" output of Fig. 7). | |
| Stage 2 (SLAT flow on the fused coords, canonical/view-agnostic): one shared | |
| noise; each view path gets its own pose-aligned conditioning (UVs projected | |
| with that view's pose, its own visual_cond + dino_cond); every Euler step | |
| uses the mean of the per-view (CFG-guided) velocities. | |
| Views are run as batch-1 forwards in a loop (memory == the 1v pipeline). | |
| With --views 1 this reduces exactly to Cupid3DPipeline.run() semantics. | |
| Usage: | |
| python batch_cupid_mv.py --selection SEL.json --inputs DIR --out OUTDIR --views 4 | |
| [--seed 42] [--shard i --nshards n] [--limit N] | |
| (--exp EXPDIR = shorthand for --selection EXPDIR/selection.json --inputs EXPDIR/inputs) | |
| Inputs: DIR/<object>_{front,side,back,oside}.png RGBA crops, alpha used as-is. | |
| Output: OUTDIR/<object>.glb (Cupid canonical frame; to_glb simplify 0.95, tex 1024, | |
| identical to batch_cupid.py / Cupid save_mesh()) | |
| OUTDIR/<object>.pose.json per-view Cupid camera (extrinsic 4x4 OpenCV | |
| world(canonical)->camera, intrinsic 3x3 in NORMALIZED image coords of the | |
| pad_to_square'd input image), plus pad/crop bookkeeping. | |
| Idempotent (skips existing .glb), per-object try/except, re-execs in cupid env. | |
| """ | |
| import os | |
| import sys | |
| ENV = "/lp-dev/jonghoon/mv-mesh/baselines/cupid/env" | |
| ENV_PY = os.path.join(ENV, "bin", "python") | |
| REPO = "/lp-dev/jonghoon/mv-mesh/baselines/cupid/repo" | |
| HF = "/lp-dev/jonghoon/mv-mesh/hf_cache" | |
| _ENV_VARS = { | |
| "PYTHONUNBUFFERED": "1", | |
| "OMP_NUM_THREADS": "4", | |
| "MKL_NUM_THREADS": "4", | |
| "SPCONV_ALGO": "native", | |
| "ATTN_BACKEND": "flash_attn", | |
| "MKL_THREADING_LAYER": "GNU", | |
| "MKL_SERVICE_FORCE_INTEL": "0", | |
| "HF_HOME": HF, | |
| "HUGGINGFACE_HUB_CACHE": HF, | |
| "HF_HUB_CACHE": HF, | |
| "TORCH_HOME": "/lp-dev/jonghoon/mv-mesh/torch_hub", | |
| "CUDA_HOME": "/usr/local/cuda-12.8", | |
| } | |
| def _reexec_in_env(): | |
| env = dict(os.environ) | |
| for k, v in _ENV_VARS.items(): | |
| env[k] = v | |
| for k in ("OMP_NUM_THREADS", "MKL_NUM_THREADS"): | |
| if os.environ.get(k): | |
| env[k] = os.environ[k] | |
| env["CONDA_PREFIX"] = ENV | |
| env["PATH"] = os.path.join(ENV, "bin") + os.pathsep + "/usr/local/cuda-12.8/bin" + os.pathsep + env.get("PATH", "") | |
| if not env.get("CUDA_VISIBLE_DEVICES"): | |
| sys.exit("ERROR: set CUDA_VISIBLE_DEVICES explicitly (shared box).") | |
| env["_CUPID_BATCH_INENV"] = "1" | |
| print(f"[mv] re-exec in {ENV_PY} (CUDA_VISIBLE_DEVICES={env.get('CUDA_VISIBLE_DEVICES')})", flush=True) | |
| os.execve(ENV_PY, [ENV_PY, os.path.abspath(__file__)] + sys.argv[1:], env) | |
| if not os.environ.get("_CUPID_BATCH_INENV"): | |
| if not os.path.isfile(ENV_PY): | |
| sys.exit(f"ERROR: env python not found: {ENV_PY}") | |
| _reexec_in_env() | |
| for _k, _v in _ENV_VARS.items(): | |
| os.environ.setdefault(_k, _v) | |
| import argparse | |
| import json | |
| import time | |
| import traceback | |
| import numpy as np | |
| os.chdir(REPO) | |
| sys.path.insert(0, REPO) | |
| VIEW_TAGS = {1: ["front"], 2: ["front", "side"], | |
| 4: ["front", "side", "back", "oside"]} | |
| def _euler_tseq(steps, rescale_t): | |
| t_seq = np.linspace(1, 0, steps + 1) | |
| t_seq = rescale_t * t_seq / (1 + (rescale_t - 1) * t_seq) | |
| return [(float(t_seq[i]), float(t_seq[i + 1])) for i in range(steps)] | |
| def run_multiview(pipeline, images, seed=42): | |
| import torch | |
| with torch.no_grad(): | |
| return _run_multiview(pipeline, images, seed) | |
| def _run_multiview(pipeline, images, seed=42): | |
| """Cupid multi-view inference (MultiDiffusion-style fusion of the shared | |
| object latent across per-view flow paths). images: list of PIL RGBA (padded | |
| to square, as sample_utils.load_image returns). Returns dict with | |
| 'mesh','gaussian' (1 fused object) and per-view 'pose' + crop params.""" | |
| import torch | |
| from cupid.modules import sparse as sp | |
| dev = pipeline.device | |
| V = len(images) | |
| torch.manual_seed(seed) | |
| # ---- per-view preprocessing + conditioning (as in run()) ---- | |
| procs, conds, visuals = [], [], [] | |
| for im in images: | |
| p = pipeline.crop_image(pipeline.preprocess_image(im)) | |
| c, vis = pipeline.get_cond([p]) # {'cond','neg_cond'}, [1,3,256,256] | |
| procs.append(p); conds.append(c); visuals.append(vis) | |
| # ---- Stage 1: sparse structure + UV (pose) flow, fused occupancy ---- | |
| fm = pipeline.models['sparse_structure_flow_model'] | |
| res, C = fm.resolution, fm.in_channels | |
| n_ss = pipeline.structure_decoder.structure_decoder.latent_channels | |
| sampler = pipeline.sparse_structure_sampler | |
| params = dict(pipeline.sparse_structure_sampler_params) | |
| noise = torch.randn(V, C, res, res, res).to(dev) # CPU RNG, as run() | |
| noise[:, :n_ss] = noise[:1, :n_ss] # one shared object (occupancy) latent | |
| x = [noise[v:v + 1].clone() for v in range(V)] | |
| for t, t_prev in _euler_tseq(params['steps'], params['rescale_t']): | |
| for v in range(V): | |
| pv = sampler._inference_model(fm, x[v], t, cond=conds[v]['cond'], | |
| neg_cond=conds[v]['neg_cond'], | |
| cfg_strength=params['cfg_strength'], | |
| cfg_interval=params['cfg_interval']) | |
| x[v] = x[v] - (t - t_prev) * pv | |
| shared = torch.stack([xv[:, :n_ss] for xv in x]).mean(0) # MultiDiffusion fuse | |
| for v in range(V): | |
| x[v][:, :n_ss] = shared | |
| z_s = torch.cat(x, 0) | |
| st = pipeline.decode_zs(z_s) # coords identical across views | |
| poses = pipeline.decode_uv(st['uvs']) # one CameraPose per view (DLT) | |
| coords0 = st['coords'][st['coords'][:, 0] == 0].clone() | |
| # ---- Stage 2: SLAT flow, one shared latent, per-view pose-aligned cond ---- | |
| sm = pipeline.models['slat_flow_model'] | |
| ssampler = pipeline.slat_sampler | |
| sparams = dict(pipeline.slat_sampler_params) | |
| feats = torch.randn(coords0.shape[0], sm.out_channels).to(dev) | |
| xs = sp.SparseTensor(feats=feats, coords=coords0) | |
| view_cond = [] | |
| for v in range(V): | |
| uvs = pipeline._make_sparse_uvs_from_pose( | |
| xs, poses[v].extrinsic[None].to(dev), poses[v].intrinsic[None].to(dev)) | |
| mc = {'visual_cond': visuals[v], 'dino_cond': conds[v]['cond']} | |
| nc = {'visual_cond': torch.zeros_like(visuals[v]), | |
| 'dino_cond': torch.zeros_like(conds[v]['cond'])} | |
| view_cond.append((uvs, mc, nc)) | |
| for t, t_prev in _euler_tseq(sparams['steps'], sparams['rescale_t']): | |
| vsum = None | |
| for uvs, mc, nc in view_cond: | |
| pv = ssampler._inference_model(sm, xs, t, cond=mc, neg_cond=nc, | |
| cfg_strength=sparams['cfg_strength'], | |
| cfg_interval=sparams['cfg_interval'], | |
| uvs=uvs) | |
| vsum = pv.feats if vsum is None else vsum + pv.feats | |
| xs = xs.replace(xs.feats - (t - t_prev) * (vsum / V)) # fused update | |
| std = torch.tensor(pipeline.slat_normalization['std'])[None].to(dev) | |
| mean = torch.tensor(pipeline.slat_normalization['mean'])[None].to(dev) | |
| slat = xs * std + mean | |
| out = pipeline.decode_slat(slat, ['mesh', 'gaussian']) | |
| out['pose'] = [po.de_crop(pr.crop_params).as_dict() for po, pr in zip(poses, procs)] | |
| out['crop_params'] = [pr.crop_params.as_tuple() for pr in procs] | |
| out['n_coords'] = int(coords0.shape[0]) | |
| return out | |
| def main(): | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--selection") | |
| ap.add_argument("--inputs") | |
| ap.add_argument("--exp", help="EXPDIR with selection.json + inputs/") | |
| ap.add_argument("--out", required=True) | |
| ap.add_argument("--views", type=int, default=4, choices=sorted(VIEW_TAGS)) | |
| ap.add_argument("--seed", type=int, default=42) | |
| ap.add_argument("--limit", type=int, default=None) | |
| ap.add_argument("--simplify", type=float, default=0.95) | |
| ap.add_argument("--texture-size", type=int, default=1024) | |
| ap.add_argument("--shard", type=int, default=0) | |
| ap.add_argument("--nshards", type=int, default=1) | |
| ap.add_argument("--reverse", action="store_true") | |
| args = ap.parse_args() | |
| if args.exp: | |
| args.selection = args.selection or os.path.join(args.exp, "selection.json") | |
| args.inputs = args.inputs or os.path.join(args.exp, "inputs") | |
| if not (args.selection and args.inputs): | |
| ap.error("need --exp or --selection + --inputs") | |
| with open(args.selection) as f: | |
| sel = json.load(f) | |
| sel = sel["selections"] if isinstance(sel, dict) else sel | |
| objects = [s["object"] for s in sel] | |
| if args.limit: | |
| objects = objects[: args.limit] | |
| if args.nshards > 1: | |
| objects = objects[args.shard::args.nshards] | |
| if args.reverse: | |
| objects = objects[::-1] | |
| tags = VIEW_TAGS[args.views] | |
| outdir = os.path.abspath(args.out) | |
| os.makedirs(outdir, exist_ok=True) | |
| todo = [o for o in objects if not os.path.isfile(os.path.join(outdir, f"{o}.glb"))] | |
| print(f"[mv] {len(objects)} object(s), {len(objects)-len(todo)} done, {len(todo)} to run " | |
| f"| views={args.views} tags={tags} seed={args.seed}", flush=True) | |
| import torch | |
| from PIL import Image | |
| from cupid.pipelines import Cupid3DPipeline | |
| from cupid.utils import sample_utils, postprocessing_utils | |
| n_ok = n_fail = n_skip = 0 | |
| pipeline = None | |
| for i, name in enumerate(objects, 1): | |
| out = os.path.join(outdir, f"{name}.glb") | |
| t0 = time.time() | |
| try: | |
| if os.path.isfile(out): | |
| n_skip += 1 | |
| print(f"[{i}/{len(objects)}] {name} SKIP", flush=True) | |
| continue | |
| paths = [os.path.join(args.inputs, f"{name}_{t}.png") for t in tags] | |
| for p in paths: | |
| if not os.path.isfile(p): | |
| raise FileNotFoundError(p) | |
| if pipeline is None: | |
| tl = time.time() | |
| pipeline = Cupid3DPipeline.from_pretrained("hbb1/Cupid") | |
| pipeline.cuda() | |
| print(f"[mv] pipeline ready in {time.time()-tl:.1f}s", flush=True) | |
| t0 = time.time() | |
| torch.cuda.reset_peak_memory_stats() | |
| raw_sizes = [Image.open(p).size for p in paths] | |
| images = [sample_utils.load_image(p) for p in paths] # pad_to_square | |
| res = run_multiview(pipeline, images, seed=args.seed) | |
| glb = postprocessing_utils.to_glb(res["gaussian"][0], res["mesh"][0], | |
| simplify=args.simplify, | |
| texture_size=args.texture_size, verbose=False) | |
| views = [] | |
| for t, p, (W, H), im, pose, cp in zip(tags, paths, raw_sizes, images, | |
| res["pose"], res["crop_params"]): | |
| S = im.size[0] | |
| views.append({ | |
| "tag": t, "image": p, "raw_size_wh": [W, H], "padded_size": S, | |
| "pad_offset_xy": [(S - W) // 2, (S - H) // 2], | |
| "cupid_crop_params": {"fov_scale": cp[0], "cx_offset": cp[1], "cy_offset": cp[2]}, | |
| "extrinsic": pose["extrinsic"].detach().cpu().numpy().tolist(), | |
| "intrinsic_normalized": pose["intrinsic"].detach().cpu().numpy().tolist(), | |
| }) | |
| pose_json = { | |
| "object": name, "views": args.views, "seed": args.seed, | |
| "method": "cupid multi-view (MultiDiffusion fusion of shared object latent; paper Fig.7)", | |
| "frame_note": ("extrinsic = OpenCV world->camera, world = Cupid canonical voxel frame " | |
| "(xyz in [-0.5,0.5]^3, the frame of Cupid's decoded mesh BEFORE to_glb's " | |
| "Y-up conversion); intrinsic maps to normalized [0,1] uv of the " | |
| "pad_to_square'd input image (multiply row0/1 by padded_size for pixels)."), | |
| "n_coords": res["n_coords"], | |
| "per_view": views, | |
| } | |
| tmp = out + ".tmp.glb" | |
| glb.export(tmp) | |
| with open(out[:-4] + ".pose.json", "w") as f: | |
| json.dump(pose_json, f, indent=1) | |
| os.replace(tmp, out) | |
| peak = torch.cuda.max_memory_allocated() / 2**30 | |
| del res, glb, images | |
| n_ok += 1 | |
| print(f"[{i}/{len(objects)}] {name} OK {time.time()-t0:.1f}s peak_alloc={peak:.2f}GiB -> {out}", flush=True) | |
| except Exception: | |
| n_fail += 1 | |
| traceback.print_exc() | |
| print(f"[{i}/{len(objects)}] {name} FAIL {time.time()-t0:.1f}s", flush=True) | |
| finally: | |
| try: | |
| torch.cuda.empty_cache() | |
| except Exception: | |
| pass | |
| print(f"[mv] DONE ok={n_ok} fail={n_fail} skip={n_skip}", flush=True) | |
| if __name__ == "__main__": | |
| main() | |