#!/usr/bin/env python """Benchmark driver: Hunyuan3D-2mv (official multi-view shape + Hunyuan3D-Paint texture). This is the TEXTURED multi-view Hunyuan baseline (replaces Hunyuan3D-Omni as "the Hunyuan baseline"). It calls the OFFICIAL pipelines exactly as repo/examples/textured_shape_gen_multiview.py does: pipeline = Hunyuan3DDiTFlowMatchingPipeline.from_pretrained( 'tencent/Hunyuan3D-2mv', subfolder='hunyuan3d-dit-v2-mv', variant='fp16') mesh = pipeline(image=, num_inference_steps=50, octree_resolution=380, num_chunks=20000, generator=torch.manual_seed(seed), output_type='trimesh')[0] paint = Hunyuan3DPaintPipeline.from_pretrained('tencent/Hunyuan3D-2', subfolder=) mesh = paint(mesh, image=) Differences from repo/examples (all deliberate, all logged in /lp-dev/jonghoon/mv-mesh/.debug_hy2mv/LOG.md section A): * seed = 42 (instead of the example's 12345). * --no-remove-bg semantics: our inputs are already RGBA cutouts with a real alpha channel, so rembg is skipped (the official example also skips rembg when the image is not RGB). NO resize/crop is applied here; the official MVImageProcessorV2 (border_ratio=0.15, 512) inside the pipeline does all geometric preprocessing. * --simplify-faces 40000 quadric decimation of the shape mesh BEFORE painting (xatlas UV-unwrap is otherwise pathologically slow on dense 380^3 MC meshes). * both pipelines are loaded ONCE and reused for every object in the shard. VIEW MAPPING (our tag -> Hunyuan key), see LOG.md A4 for the evidence: front -> 'front', side -> 'right', back -> 'back', oside -> 'left' Hunyuan3D-2mv supports AT MOST 4 views (MVImageProcessorV2.view2idx has 4 slots), so --views 8 is rejected. OUTPUT FRAME: Hunyuan's own generation frame -> must be run through metrics/align_baselines.py before metrics/appeval/evaluate_appforce.py. CLI --- python batch_hy3d_2mv.py --selection SEL.json --inputs INPUTDIR --out OUTDIR --views {1|2|4} [--seed 42] [--gpu G] [--shard i --nshards n] [--limit N] [--no-texture] [--simplify-faces 40000] [--steps 50] [--octree-resolution 380] Idempotent (skips an existing OUTDIR/.glb), atomic write, per-object try/except with an OOM retry, per-object timing printed to the log. """ import os import sys HY_ROOT = "/lp-dev/jonghoon/mv-mesh/hunyuan3d-2mv" REPO = os.path.join(HY_ROOT, "repo") ENV = os.path.join(HY_ROOT, "env") ENV_PY = os.path.join(ENV, "bin", "python") HF = "/lp-dev/jonghoon/mv-mesh/hf_cache" _ENV_VARS = { "PYTHONUNBUFFERED": "1", "OMP_NUM_THREADS": "4", "MKL_NUM_THREADS": "4", "SPCONV_ALGO": "native", "PYOPENGL_PLATFORM": "egl", "PYTORCH_CUDA_ALLOC_CONF": "expandable_segments:True", "HF_HOME": HF, "HUGGINGFACE_HUB_CACHE": HF, "HF_HUB_CACHE": HF, "U2NET_HOME": os.path.join(HF, "u2net"), } def _argv_get(flag): for i, a in enumerate(sys.argv): if a == flag and i + 1 < len(sys.argv): return sys.argv[i + 1] if a.startswith(flag + "="): return a.split("=", 1)[1] return None def _reexec_in_env(): env = dict(os.environ) env.update(_ENV_VARS) env["CONDA_PREFIX"] = ENV env["PATH"] = os.path.join(ENV, "bin") + os.pathsep + env.get("PATH", "") env["PYTHONPATH"] = REPO + os.pathsep + env.get("PYTHONPATH", "") if not env.get("CUDA_VISIBLE_DEVICES"): g = _argv_get("--gpu") if g: env["CUDA_VISIBLE_DEVICES"] = g env["_HY3D2MV_BATCH_INENV"] = "1" print(f"[hy2mv] re-exec in {ENV_PY} (CUDA={env.get('CUDA_VISIBLE_DEVICES')})", flush=True) os.execve(ENV_PY, [ENV_PY, os.path.abspath(__file__)] + sys.argv[1:], env) if not os.environ.get("_HY3D2MV_BATCH_INENV"): if not os.path.isfile(ENV_PY): sys.exit(f"ERROR: env python not found: {ENV_PY}") _reexec_in_env() for _k, _v in _ENV_VARS.items(): os.environ.setdefault(_k, _v) os.makedirs(os.environ["U2NET_HOME"], exist_ok=True) import argparse # noqa: E402 import gc # noqa: E402 import json # noqa: E402 import time # noqa: E402 import traceback # noqa: E402 from pathlib import Path # noqa: E402 _ORIG_CWD = os.getcwd() sys.path.insert(0, REPO) os.chdir(REPO) def _abs(p): """abspath relative to the ORIGINAL cwd (we chdir into the repo above).""" return p if os.path.isabs(p) else os.path.normpath(os.path.join(_ORIG_CWD, p)) import torch # noqa: E402 from PIL import Image # noqa: E402 # our render tag -> official Hunyuan3D-2mv view key (see LOG.md A4) TAG2VIEW = {"front": "front", "side": "right", "back": "back", "oside": "left"} # negative control only: HY2MV_SWAP_SIDE=1 uses the WRONG (mirrored) mapping if os.environ.get("HY2MV_SWAP_SIDE") == "1": TAG2VIEW = {"front": "front", "side": "left", "back": "back", "oside": "right"} VIEW_TAGS = {1: ["front"], 2: ["front", "side"], 4: ["front", "side", "back", "oside"]} def load_views(inputs_dir, obj, tags, remove_bg, rembg): """{hunyuan_key: PIL RGBA} built BY TAG (never positionally).""" d, srcs = {}, {} for tag in tags: p = os.path.join(inputs_dir, f"{obj}_{tag}.png") if not os.path.isfile(p): raise FileNotFoundError(p) img = Image.open(p) has_alpha = img.mode in ("RGBA", "LA") or ( img.mode == "P" and "transparency" in img.info) if remove_bg or not has_alpha: img = rembg(img.convert("RGB")) else: img = img.convert("RGBA") key = TAG2VIEW[tag] d[key] = img srcs[key] = os.path.basename(p) return d, srcs def main(): ap = argparse.ArgumentParser( description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument("--selection", required=True) ap.add_argument("--inputs", required=True) ap.add_argument("--exp", default=None, help="unused; CLI parity with the other drivers") ap.add_argument("--out", required=True) ap.add_argument("--views", type=int, choices=[1, 2, 4], required=True) ap.add_argument("--seed", type=int, default=42) ap.add_argument("--steps", type=int, default=50) ap.add_argument("--octree-resolution", type=int, default=380) ap.add_argument("--num-chunks", type=int, default=20000) ap.add_argument("--no-texture", action="store_true") ap.add_argument("--remove-bg", action="store_true", help="run rembg even on RGBA inputs (default: off, like --no-remove-bg)") ap.add_argument("--paint-subfolder", default="hunyuan3d-paint-v2-0-turbo") ap.add_argument("--simplify-faces", type=int, default=40000) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--objects", nargs="+", default=None) ap.add_argument("--gpu", default=None) ap.add_argument("--shard", type=int, default=0) ap.add_argument("--nshards", type=int, default=1) args = ap.parse_args() inputs_dir = _abs(args.inputs) out_dir = _abs(args.out) args.selection = _abs(args.selection) os.makedirs(out_dir, exist_ok=True) tags = VIEW_TAGS[args.views] data = json.loads(Path(args.selection).read_text()) sels = data["selections"] if isinstance(data, dict) else data objects = [s["object"] for s in sels] if args.objects: want = set(args.objects) objects = [o for o in objects if o in want] if args.limit: objects = objects[: args.limit] if args.nshards > 1: objects = objects[args.shard::args.nshards] print(f"[hy2mv] views={args.views} tags={tags} -> keys={[TAG2VIEW[t] for t in tags]} " f"seed={args.seed} steps={args.steps} octree={args.octree_resolution} " f"texture={not args.no_texture} paint={args.paint_subfolder} " f"simplify={args.simplify_faces} shard={args.shard}/{args.nshards} " f"n_obj={len(objects)} CUDA={os.environ.get('CUDA_VISIBLE_DEVICES')}", flush=True) todo = [o for o in objects if not os.path.isfile(os.path.join(out_dir, f"{o}.glb"))] print(f"[hy2mv] {len(objects) - len(todo)} already present, {len(todo)} to do", flush=True) if not todo: print("HY3D2MV DONE ok=0 fail=0 skip=%d total=%d" % (len(objects), len(objects)), flush=True) return 0 rembg = None if args.remove_bg: from hy3dgen.rembg import BackgroundRemover rembg = BackgroundRemover() from hy3dgen.shapegen import Hunyuan3DDiTFlowMatchingPipeline t0 = time.time() print("[hy2mv] loading shape pipeline tencent/Hunyuan3D-2mv:hunyuan3d-dit-v2-mv ...", flush=True) shape_pipe = Hunyuan3DDiTFlowMatchingPipeline.from_pretrained( "tencent/Hunyuan3D-2mv", subfolder="hunyuan3d-dit-v2-mv", variant="fp16") print(f"[hy2mv] shape pipeline ready in {time.time() - t0:.1f}s", flush=True) paint_pipe = None if not args.no_texture: from hy3dgen.texgen import Hunyuan3DPaintPipeline t0 = time.time() print(f"[hy2mv] loading paint pipeline tencent/Hunyuan3D-2:{args.paint_subfolder} ...", flush=True) paint_pipe = Hunyuan3DPaintPipeline.from_pretrained( "tencent/Hunyuan3D-2", subfolder=args.paint_subfolder) print(f"[hy2mv] paint pipeline ready in {time.time() - t0:.1f}s", flush=True) def gen_one(obj, octree, simplify): image_dict, srcs = load_views(inputs_dir, obj, tags, remove_bg=args.remove_bg, rembg=rembg) mesh = shape_pipe( image=image_dict, num_inference_steps=args.steps, octree_resolution=octree, num_chunks=args.num_chunks, generator=torch.manual_seed(args.seed), output_type="trimesh", )[0] nf_raw = len(mesh.faces) textured = False if paint_pipe is not None: if simplify and len(mesh.faces) > simplify: try: s = mesh.simplify_quadric_decimation(face_count=simplify) if s is not None and len(s.faces) > 0: mesh = s except Exception as e: print(f"[hy2mv] simplify failed ({e}); using full mesh", flush=True) mesh = paint_pipe(mesh, image=image_dict["front"]) textured = True return mesh, srcs, nf_raw, textured counts = {"ok": 0, "fail": 0, "skip": len(objects) - len(todo)} times = [] for i, obj in enumerate(todo, 1): out_path = os.path.join(out_dir, f"{obj}.glb") if os.path.isfile(out_path): # another shard/lane may have made it counts["skip"] += 1 continue t1 = time.time() mesh = None for attempt, (octree, simplify) in enumerate( [(args.octree_resolution, args.simplify_faces), (max(256, args.octree_resolution - 124), min(args.simplify_faces, 20000))]): try: mesh, srcs, nf_raw, textured = gen_one(obj, octree, simplify) break except torch.cuda.OutOfMemoryError: print(f"[hy2mv {i}/{len(todo)}] OOM {obj} (attempt {attempt + 1}), retrying smaller", flush=True) mesh = None gc.collect(); torch.cuda.empty_cache() except Exception: traceback.print_exc() mesh = None gc.collect(); torch.cuda.empty_cache() break if mesh is None or len(getattr(mesh, "vertices", [])) == 0: counts["fail"] += 1 print(f"[hy2mv {i}/{len(todo)}] FAIL {obj} {time.time() - t1:.1f}s", flush=True) continue try: tmp = out_path + f".tmp{os.getpid()}.glb" mesh.export(tmp) os.replace(tmp, out_path) Path(out_path[:-4] + ".views.json").write_text(json.dumps( {"views": args.views, "tag2view": {t: TAG2VIEW[t] for t in tags}, "sources": srcs, "seed": args.seed, "steps": args.steps, "octree_resolution": octree, "simplify_faces": simplify, "raw_faces": int(nf_raw), "textured": bool(textured), "paint_subfolder": args.paint_subfolder if textured else None}, indent=2)) dt = time.time() - t1 times.append(dt) counts["ok"] += 1 print(f"[hy2mv {i}/{len(todo)}] OK {obj} {dt:.1f}s verts={len(mesh.vertices)} " f"faces={len(mesh.faces)} raw_faces={nf_raw} tex={textured} " f"({os.path.getsize(out_path)} bytes)", flush=True) except Exception: counts["fail"] += 1 traceback.print_exc() print(f"[hy2mv {i}/{len(todo)}] EXPORT FAIL {obj}", flush=True) del mesh gc.collect(); torch.cuda.empty_cache() avg = sum(times) / len(times) if times else 0.0 print(f"HY3D2MV DONE ok={counts['ok']} fail={counts['fail']} skip={counts['skip']} " f"total={len(objects)} avg_s_per_obj={avg:.1f}", flush=True) return 1 if counts["fail"] else 0 if __name__ == "__main__": sys.exit(main())