bert_simpson / forgebench /code /baselines /batch_hy3d_2mv.py
Ronaldo-GOAT's picture
Add files using upload-large-folder tool
4c3d957 verified
Raw History Blame Contribute Delete
13.3 kB
#!/usr/bin/env python
"""Benchmark driver: Hunyuan3D-2mv (official multi-view shape + Hunyuan3D-Paint texture).
This is the TEXTURED multi-view Hunyuan baseline (replaces Hunyuan3D-Omni as
"the Hunyuan baseline"). It calls the OFFICIAL pipelines exactly as
repo/examples/textured_shape_gen_multiview.py does:
pipeline = Hunyuan3DDiTFlowMatchingPipeline.from_pretrained(
'tencent/Hunyuan3D-2mv', subfolder='hunyuan3d-dit-v2-mv', variant='fp16')
mesh = pipeline(image=<dict of views>, num_inference_steps=50,
octree_resolution=380, num_chunks=20000,
generator=torch.manual_seed(seed), output_type='trimesh')[0]
paint = Hunyuan3DPaintPipeline.from_pretrained('tencent/Hunyuan3D-2',
subfolder=<paint subfolder>)
mesh = paint(mesh, image=<front view RGBA>)
Differences from repo/examples (all deliberate, all logged in
/lp-dev/jonghoon/mv-mesh/.debug_hy2mv/LOG.md section A):
* seed = 42 (instead of the example's 12345).
* --no-remove-bg semantics: our inputs are already RGBA cutouts with a real
alpha channel, so rembg is skipped (the official example also skips rembg
when the image is not RGB). NO resize/crop is applied here; the official
MVImageProcessorV2 (border_ratio=0.15, 512) inside the pipeline does all
geometric preprocessing.
* --simplify-faces 40000 quadric decimation of the shape mesh BEFORE painting
(xatlas UV-unwrap is otherwise pathologically slow on dense 380^3 MC meshes).
* both pipelines are loaded ONCE and reused for every object in the shard.
VIEW MAPPING (our tag -> Hunyuan key), see LOG.md A4 for the evidence:
front -> 'front', side -> 'right', back -> 'back', oside -> 'left'
Hunyuan3D-2mv supports AT MOST 4 views (MVImageProcessorV2.view2idx has 4 slots),
so --views 8 is rejected.
OUTPUT FRAME: Hunyuan's own generation frame -> must be run through
metrics/align_baselines.py before metrics/appeval/evaluate_appforce.py.
CLI
---
python batch_hy3d_2mv.py --selection SEL.json --inputs INPUTDIR --out OUTDIR
--views {1|2|4} [--seed 42] [--gpu G] [--shard i --nshards n] [--limit N]
[--no-texture] [--simplify-faces 40000] [--steps 50] [--octree-resolution 380]
Idempotent (skips an existing OUTDIR/<obj>.glb), atomic write, per-object
try/except with an OOM retry, per-object timing printed to the log.
"""
import os
import sys
HY_ROOT = "/lp-dev/jonghoon/mv-mesh/hunyuan3d-2mv"
REPO = os.path.join(HY_ROOT, "repo")
ENV = os.path.join(HY_ROOT, "env")
ENV_PY = os.path.join(ENV, "bin", "python")
HF = "/lp-dev/jonghoon/mv-mesh/hf_cache"
_ENV_VARS = {
"PYTHONUNBUFFERED": "1",
"OMP_NUM_THREADS": "4",
"MKL_NUM_THREADS": "4",
"SPCONV_ALGO": "native",
"PYOPENGL_PLATFORM": "egl",
"PYTORCH_CUDA_ALLOC_CONF": "expandable_segments:True",
"HF_HOME": HF,
"HUGGINGFACE_HUB_CACHE": HF,
"HF_HUB_CACHE": HF,
"U2NET_HOME": os.path.join(HF, "u2net"),
}
def _argv_get(flag):
for i, a in enumerate(sys.argv):
if a == flag and i + 1 < len(sys.argv):
return sys.argv[i + 1]
if a.startswith(flag + "="):
return a.split("=", 1)[1]
return None
def _reexec_in_env():
env = dict(os.environ)
env.update(_ENV_VARS)
env["CONDA_PREFIX"] = ENV
env["PATH"] = os.path.join(ENV, "bin") + os.pathsep + env.get("PATH", "")
env["PYTHONPATH"] = REPO + os.pathsep + env.get("PYTHONPATH", "")
if not env.get("CUDA_VISIBLE_DEVICES"):
g = _argv_get("--gpu")
if g:
env["CUDA_VISIBLE_DEVICES"] = g
env["_HY3D2MV_BATCH_INENV"] = "1"
print(f"[hy2mv] re-exec in {ENV_PY} (CUDA={env.get('CUDA_VISIBLE_DEVICES')})", flush=True)
os.execve(ENV_PY, [ENV_PY, os.path.abspath(__file__)] + sys.argv[1:], env)
if not os.environ.get("_HY3D2MV_BATCH_INENV"):
if not os.path.isfile(ENV_PY):
sys.exit(f"ERROR: env python not found: {ENV_PY}")
_reexec_in_env()
for _k, _v in _ENV_VARS.items():
os.environ.setdefault(_k, _v)
os.makedirs(os.environ["U2NET_HOME"], exist_ok=True)
import argparse # noqa: E402
import gc # noqa: E402
import json # noqa: E402
import time # noqa: E402
import traceback # noqa: E402
from pathlib import Path # noqa: E402
_ORIG_CWD = os.getcwd()
sys.path.insert(0, REPO)
os.chdir(REPO)
def _abs(p):
"""abspath relative to the ORIGINAL cwd (we chdir into the repo above)."""
return p if os.path.isabs(p) else os.path.normpath(os.path.join(_ORIG_CWD, p))
import torch # noqa: E402
from PIL import Image # noqa: E402
# our render tag -> official Hunyuan3D-2mv view key (see LOG.md A4)
TAG2VIEW = {"front": "front", "side": "right", "back": "back", "oside": "left"}
# negative control only: HY2MV_SWAP_SIDE=1 uses the WRONG (mirrored) mapping
if os.environ.get("HY2MV_SWAP_SIDE") == "1":
TAG2VIEW = {"front": "front", "side": "left", "back": "back", "oside": "right"}
VIEW_TAGS = {1: ["front"],
2: ["front", "side"],
4: ["front", "side", "back", "oside"]}
def load_views(inputs_dir, obj, tags, remove_bg, rembg):
"""{hunyuan_key: PIL RGBA} built BY TAG (never positionally)."""
d, srcs = {}, {}
for tag in tags:
p = os.path.join(inputs_dir, f"{obj}_{tag}.png")
if not os.path.isfile(p):
raise FileNotFoundError(p)
img = Image.open(p)
has_alpha = img.mode in ("RGBA", "LA") or (
img.mode == "P" and "transparency" in img.info)
if remove_bg or not has_alpha:
img = rembg(img.convert("RGB"))
else:
img = img.convert("RGBA")
key = TAG2VIEW[tag]
d[key] = img
srcs[key] = os.path.basename(p)
return d, srcs
def main():
ap = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--selection", required=True)
ap.add_argument("--inputs", required=True)
ap.add_argument("--exp", default=None, help="unused; CLI parity with the other drivers")
ap.add_argument("--out", required=True)
ap.add_argument("--views", type=int, choices=[1, 2, 4], required=True)
ap.add_argument("--seed", type=int, default=42)
ap.add_argument("--steps", type=int, default=50)
ap.add_argument("--octree-resolution", type=int, default=380)
ap.add_argument("--num-chunks", type=int, default=20000)
ap.add_argument("--no-texture", action="store_true")
ap.add_argument("--remove-bg", action="store_true",
help="run rembg even on RGBA inputs (default: off, like --no-remove-bg)")
ap.add_argument("--paint-subfolder", default="hunyuan3d-paint-v2-0-turbo")
ap.add_argument("--simplify-faces", type=int, default=40000)
ap.add_argument("--limit", type=int, default=0)
ap.add_argument("--objects", nargs="+", default=None)
ap.add_argument("--gpu", default=None)
ap.add_argument("--shard", type=int, default=0)
ap.add_argument("--nshards", type=int, default=1)
args = ap.parse_args()
inputs_dir = _abs(args.inputs)
out_dir = _abs(args.out)
args.selection = _abs(args.selection)
os.makedirs(out_dir, exist_ok=True)
tags = VIEW_TAGS[args.views]
data = json.loads(Path(args.selection).read_text())
sels = data["selections"] if isinstance(data, dict) else data
objects = [s["object"] for s in sels]
if args.objects:
want = set(args.objects)
objects = [o for o in objects if o in want]
if args.limit:
objects = objects[: args.limit]
if args.nshards > 1:
objects = objects[args.shard::args.nshards]
print(f"[hy2mv] views={args.views} tags={tags} -> keys={[TAG2VIEW[t] for t in tags]} "
f"seed={args.seed} steps={args.steps} octree={args.octree_resolution} "
f"texture={not args.no_texture} paint={args.paint_subfolder} "
f"simplify={args.simplify_faces} shard={args.shard}/{args.nshards} "
f"n_obj={len(objects)} CUDA={os.environ.get('CUDA_VISIBLE_DEVICES')}", flush=True)
todo = [o for o in objects if not os.path.isfile(os.path.join(out_dir, f"{o}.glb"))]
print(f"[hy2mv] {len(objects) - len(todo)} already present, {len(todo)} to do", flush=True)
if not todo:
print("HY3D2MV DONE ok=0 fail=0 skip=%d total=%d" % (len(objects), len(objects)), flush=True)
return 0
rembg = None
if args.remove_bg:
from hy3dgen.rembg import BackgroundRemover
rembg = BackgroundRemover()
from hy3dgen.shapegen import Hunyuan3DDiTFlowMatchingPipeline
t0 = time.time()
print("[hy2mv] loading shape pipeline tencent/Hunyuan3D-2mv:hunyuan3d-dit-v2-mv ...", flush=True)
shape_pipe = Hunyuan3DDiTFlowMatchingPipeline.from_pretrained(
"tencent/Hunyuan3D-2mv", subfolder="hunyuan3d-dit-v2-mv", variant="fp16")
print(f"[hy2mv] shape pipeline ready in {time.time() - t0:.1f}s", flush=True)
paint_pipe = None
if not args.no_texture:
from hy3dgen.texgen import Hunyuan3DPaintPipeline
t0 = time.time()
print(f"[hy2mv] loading paint pipeline tencent/Hunyuan3D-2:{args.paint_subfolder} ...",
flush=True)
paint_pipe = Hunyuan3DPaintPipeline.from_pretrained(
"tencent/Hunyuan3D-2", subfolder=args.paint_subfolder)
print(f"[hy2mv] paint pipeline ready in {time.time() - t0:.1f}s", flush=True)
def gen_one(obj, octree, simplify):
image_dict, srcs = load_views(inputs_dir, obj, tags,
remove_bg=args.remove_bg, rembg=rembg)
mesh = shape_pipe(
image=image_dict,
num_inference_steps=args.steps,
octree_resolution=octree,
num_chunks=args.num_chunks,
generator=torch.manual_seed(args.seed),
output_type="trimesh",
)[0]
nf_raw = len(mesh.faces)
textured = False
if paint_pipe is not None:
if simplify and len(mesh.faces) > simplify:
try:
s = mesh.simplify_quadric_decimation(face_count=simplify)
if s is not None and len(s.faces) > 0:
mesh = s
except Exception as e:
print(f"[hy2mv] simplify failed ({e}); using full mesh", flush=True)
mesh = paint_pipe(mesh, image=image_dict["front"])
textured = True
return mesh, srcs, nf_raw, textured
counts = {"ok": 0, "fail": 0, "skip": len(objects) - len(todo)}
times = []
for i, obj in enumerate(todo, 1):
out_path = os.path.join(out_dir, f"{obj}.glb")
if os.path.isfile(out_path): # another shard/lane may have made it
counts["skip"] += 1
continue
t1 = time.time()
mesh = None
for attempt, (octree, simplify) in enumerate(
[(args.octree_resolution, args.simplify_faces),
(max(256, args.octree_resolution - 124), min(args.simplify_faces, 20000))]):
try:
mesh, srcs, nf_raw, textured = gen_one(obj, octree, simplify)
break
except torch.cuda.OutOfMemoryError:
print(f"[hy2mv {i}/{len(todo)}] OOM {obj} (attempt {attempt + 1}), retrying smaller",
flush=True)
mesh = None
gc.collect(); torch.cuda.empty_cache()
except Exception:
traceback.print_exc()
mesh = None
gc.collect(); torch.cuda.empty_cache()
break
if mesh is None or len(getattr(mesh, "vertices", [])) == 0:
counts["fail"] += 1
print(f"[hy2mv {i}/{len(todo)}] FAIL {obj} {time.time() - t1:.1f}s", flush=True)
continue
try:
tmp = out_path + f".tmp{os.getpid()}.glb"
mesh.export(tmp)
os.replace(tmp, out_path)
Path(out_path[:-4] + ".views.json").write_text(json.dumps(
{"views": args.views, "tag2view": {t: TAG2VIEW[t] for t in tags},
"sources": srcs, "seed": args.seed, "steps": args.steps,
"octree_resolution": octree, "simplify_faces": simplify,
"raw_faces": int(nf_raw), "textured": bool(textured),
"paint_subfolder": args.paint_subfolder if textured else None}, indent=2))
dt = time.time() - t1
times.append(dt)
counts["ok"] += 1
print(f"[hy2mv {i}/{len(todo)}] OK {obj} {dt:.1f}s verts={len(mesh.vertices)} "
f"faces={len(mesh.faces)} raw_faces={nf_raw} tex={textured} "
f"({os.path.getsize(out_path)} bytes)", flush=True)
except Exception:
counts["fail"] += 1
traceback.print_exc()
print(f"[hy2mv {i}/{len(todo)}] EXPORT FAIL {obj}", flush=True)
del mesh
gc.collect(); torch.cuda.empty_cache()
avg = sum(times) / len(times) if times else 0.0
print(f"HY3D2MV DONE ok={counts['ok']} fail={counts['fail']} skip={counts['skip']} "
f"total={len(objects)} avg_s_per_obj={avg:.1f}", flush=True)
return 1 if counts["fail"] else 0
if __name__ == "__main__":
sys.exit(main())