Spark-H3 / scripts /benchmark_current_warmup_diffusers_comfyui_2gpu_20261001.py
Aazeus's picture
Publish Spark-H3 code and model card (part 2)
22a73d3 verified
Raw History Blame Contribute Delete
24.6 kB
#!/usr/bin/env python3
"""Measure current Dense versus Spark-10 on Diffusers and ComfyUI.
GPU 0 runs the compiled Diffusers BF16 stack. GPU 1 runs the native ComfyUI
INT8-convrot stack. Both use one shared cached conditioning, 20 requested
steps, Spark's 20% dense schedule warmup, one permanently dense transformer
layer, one excluded full generation per method and shape, and two measured
generations.
"""
from __future__ import annotations
import contextlib
import dataclasses
import hashlib
import json
import os
from pathlib import Path
import shutil
import statistics
import subprocess
import sys
import time
NAME = "current_warmup_diffusers_comfyui_2gpu_20261001_v2"
REPO = Path(__file__).resolve().parents[1]
BENCH = REPO.parent / "MiniMax-H3-Benchmark"
COMFY = REPO.parent / "ComfyUI"
ROOT = Path("/autodl-fs/data/h3_experiments") / NAME
OUT = Path("/autodl-fs/data/h3_outputs") / NAME
DIFF_OUT = OUT / "diffusers"
COMFY_OUT = OUT / "comfyui"
PYTHON = Path("/root/miniconda3/bin/python")
SAMPLES = BENCH / "vbench_core5_percent_subsets/20pct/samples.json"
CONDITIONING_ROOT = Path(
"/autodl-fs/data/h3_outputs/vbench20pct_768p10s_seed42_20260913"
)
CONDITIONING_FILE = CONDITIONING_ROOT / "conditioning_cache" / (
"5f6378bba24cd7e3ee647fa1e6dfea3c42acd5f53b30a577d7084242204ac362.pt"
)
EMBEDDING_DIR = COMFY / "models" / "embeddings" / NAME
EMBEDDING_FILE = EMBEDDING_DIR / "shared_case.safetensors"
EMBEDDING_NAME = f"{NAME}/shared_case.safetensors"
BASE_MODEL = "minimax_h3_fl2va_pruned_int8_convrot.safetensors"
WIDTH, HEIGHT, STEPS, SEED = 1344, 768, 20, 42
MEASURED_SEEDS = (42, 43)
DURATIONS = {
"5s": {"diffusers_frames": 120, "model_frames": 124},
"10s": {"diffusers_frames": 240, "model_frames": 243},
"14p4s": {"diffusers_frames": 345, "model_frames": 345},
}
METHODS = ("dense", "spark_10pct")
PORT = 8731
def write_json(path: Path, value: object) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
temporary = path.with_name(f"{path.name}.tmp-{os.getpid()}")
temporary.write_text(
json.dumps(value, indent=2, ensure_ascii=False, default=str) + "\n"
)
temporary.replace(path)
def sha256(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as handle:
for block in iter(lambda: handle.read(8 << 20), b""):
digest.update(block)
return digest.hexdigest()
def git_output(*args: str) -> str:
return subprocess.check_output(
["git", "-C", str(REPO), *args], text=True
).strip()
def prepare() -> None:
if ROOT.exists() or OUT.exists():
raise FileExistsError(f"refusing to overwrite {ROOT} or {OUT}")
for path in (
ROOT / "diffusers" / "records",
ROOT / "comfyui" / "records",
DIFF_OUT,
COMFY_OUT,
):
path.mkdir(parents=True, exist_ok=True)
for name in ("conditioning_cache", "conditioning_manifest.json"):
source = CONDITIONING_ROOT / name
(DIFF_OUT / name).symlink_to(
source, target_is_directory=source.is_dir()
)
import safetensors.torch
import torch
cached = torch.load(CONDITIONING_FILE, map_location="cpu", weights_only=False)
values = cached["values"]
EMBEDDING_DIR.mkdir(parents=True, exist_ok=True)
safetensors.torch.save_file(
{
"conditioning": values["prompt_embeds"].contiguous(),
"minimax_token_tags": values["text_token_tags"].contiguous(),
},
str(EMBEDDING_FILE),
metadata={"conditioning_options": "{}"},
)
source_files = [
REPO / "h3_sparse_attention" / "processor.py",
REPO / "h3_sparse_attention" / "spark_integration.py",
REPO / "h3_sparse_attention" / "global_weighted_route.py",
REPO / "comfyui_nodes.py",
REPO / "comfyui_backend.py",
COMFY / "comfy_extras" / "nodes_sparse_attention.py",
BENCH / "scripts" / "minimax_h3_vbench_4gpu_pipeline.py",
Path(__file__).resolve(),
]
config = diffusers_spark_config()
protocol = {
"status": "running",
"name": NAME,
"purpose": (
"Current-stack 5s/10s/14.4s Dense versus Spark-H3 Top-K 10% "
"timing on Diffusers and ComfyUI"
),
"started_unix": time.time(),
"git_revision": git_output("rev-parse", "HEAD"),
"git_status": git_output("status", "--short"),
"git_diff_sha256": hashlib.sha256(
subprocess.check_output(["git", "-C", str(REPO), "diff", "--binary"])
).hexdigest(),
"source_sha256": {str(path): sha256(path) for path in source_files},
"conditioning": {
"source": str(CONDITIONING_FILE),
"source_sha256": sha256(CONDITIONING_FILE),
"converted": str(EMBEDDING_FILE),
"converted_sha256": sha256(EMBEDDING_FILE),
"prompt_sha256": hashlib.sha256(cached["prompt"].encode()).hexdigest(),
"embedding_shape": list(values["prompt_embeds"].shape),
"embedding_dtype": str(values["prompt_embeds"].dtype),
},
"hardware": {
"diffusers_physical_gpu": 0,
"comfyui_physical_gpu": 1,
"required": "NVIDIA RTX PRO 6000 Blackwell, SM120",
},
"shared": {
"resolution": [WIDTH, HEIGHT],
"requested_steps": STEPS,
"seed": SEED,
"measured_seeds": list(MEASURED_SEEDS),
"durations": DURATIONS,
"methods": list(METHODS),
"runtime_warmup": (
"one excluded full generation for every backend, duration, and method"
),
"timed_repeats": len(MEASURED_SEEDS),
},
"diffusers": {
"model": "/autodl-fs/data/models/MiniMax-H3",
"dtype": "bfloat16",
"torch_compile": True,
"requested_steps": 20,
"actual_transformer_evaluations": 19,
"spark_dense_schedule_evaluations": config.dense_evaluations,
"spark_config": dataclasses.asdict(config),
"timing": "CUDA-synchronized denoising pipeline only",
},
"comfyui": {
"model": BASE_MODEL,
"model_path": str(
Path("/autodl-fs/data/models/ComfyUI-MiniMax-H3/diffusion_models")
/ BASE_MODEL
),
"requested_steps": 20,
"actual_model_evaluations": 20,
"spark_dense_schedule_evaluations": 4,
"spark": {
"warmup_mode": "warmup_ratio",
"warmup_ratio": 0.2,
"warmup_steps": 4,
"topk_mode": "topk_ratio",
"topk_ratio": 0.1,
"dense_layers": 1,
"min_tokens": 12288,
"tail_granularity": "query",
"global_anchor_dtype": "float32 (node default)",
"midpoint_direction_mode": "fused (node default)",
},
"timing": "SamplerCustomAdvanced node wall time only",
},
}
write_json(ROOT / "protocol.json", protocol)
shutil.copy2(__file__, ROOT / "runner_source.py")
def pipeline_module():
scripts = BENCH / "scripts"
if str(scripts) not in sys.path:
sys.path.insert(0, str(scripts))
import _impl_bootstrap # noqa: F401
import minimax_h3_vbench_4gpu_pipeline
return minimax_h3_vbench_4gpu_pipeline
def diffusers_args():
return pipeline_module().build_parser().parse_args(
[
"denoise",
"--samples",
str(SAMPLES),
"--output",
str(DIFF_OUT),
"--method",
"dense",
"--case-indices",
"1",
"--steps",
str(STEPS),
"--frames",
"120",
"--height",
str(HEIGHT),
"--width",
str(WIDTH),
"--workers",
"1",
]
)
def diffusers_spark_config():
if str(REPO) not in sys.path:
sys.path.insert(0, str(REPO))
from h3_sparse_attention import H3SparseAttentionConfig
return H3SparseAttentionConfig.spark(
STEPS,
warmup_percent=20.0,
sol_dense_layers=1,
sol_route_topk_ratio=0.10,
sol_log_density=False,
)
def run_diffusers_once(pipe, state, *, frames: int, seed: int, plugin=None) -> dict:
import torch
p = pipeline_module()
inference_state = p.clone_state(state)
inference_state.values["prompt_embeds"] = inference_state.values[
"prompt_embeds"
].to("cuda")
if plugin is not None:
plugin.reset()
torch.cuda.reset_peak_memory_stats()
torch.cuda.synchronize()
started = time.perf_counter()
with torch.inference_mode():
output = pipe(
state=inference_state,
num_frames=frames,
height=HEIGHT,
width=WIDTH,
num_inference_steps=STEPS,
generator=torch.Generator(device="cpu").manual_seed(seed),
output=["latents", "audio_latents"],
)
torch.cuda.synchronize()
seconds = time.perf_counter() - started
finite = all(
torch.isfinite(output[name]).all().item()
for name in ("latents", "audio_latents")
)
summary = None if plugin is None else plugin.summary()
record = {
"seconds": seconds,
"seed": seed,
"finite": finite,
"peak_memory_gib": torch.cuda.max_memory_allocated() / 2**30,
"attention_summary": summary,
}
del output, inference_state
return record
def diffusers_worker() -> None:
import torch
from h3_sparse_attention import install_h3_sparse_attention
p = pipeline_module()
args = diffusers_args()
cases = p.load_cases(SAMPLES, [1], expected_indices=tuple(range(1, 51)))
workflow, states = p.configure_denoise_workflow(args, cases)
loaded_at = time.perf_counter()
pipe, manager, acceleration, placement = p.load_denoiser(args, workflow)
load_seconds = time.perf_counter() - loaded_at
metadata = {
"gpu": torch.cuda.get_device_name(),
"capability": list(torch.cuda.get_device_capability()),
"placement": placement,
"load_seconds": load_seconds,
}
write_json(ROOT / "diffusers" / "worker.json", metadata)
try:
for duration_index, (duration, shape) in enumerate(DURATIONS.items()):
method_order = METHODS if duration_index % 2 == 0 else METHODS[::-1]
for method in method_order:
config = diffusers_spark_config() if method == "spark_10pct" else None
context = (
install_h3_sparse_attention(pipe.transformer, config)
if config is not None
else contextlib.nullcontext(None)
)
with context as plugin:
for phase, repeat, seed in (
("warmup", 0, 41),
("measured", 1, MEASURED_SEEDS[0]),
("measured", 2, MEASURED_SEEDS[1]),
):
result = run_diffusers_once(
pipe,
states[0],
frames=shape["diffusers_frames"],
seed=seed,
plugin=plugin,
)
summary = result["attention_summary"]
if summary is not None:
assert summary["completed_evaluations"] == 19, summary
assert summary["dense_evaluations"] == 4, summary
record = {
"backend": "diffusers",
"duration": duration,
"method": method,
"phase": phase,
"repeat": repeat,
"requested_frames": shape["diffusers_frames"],
"model_frames": shape["model_frames"],
**result,
}
target = ROOT / "diffusers" / "records" / (
f"{duration}_{method}_{phase}_{repeat}.json"
)
write_json(target, record)
print(
"DIFFUSERS",
duration,
method,
phase,
repeat,
f"{result['seconds']:.3f}s",
flush=True,
)
finally:
acceleration.remove()
del pipe, manager
p.release_cpu_arenas()
def comfy_graph(method: str, frames: int, seed: int, target: Path) -> dict:
nodes = {
"1": {
"class_type": "UNETLoader",
"inputs": {"unet_name": BASE_MODEL, "weight_dtype": "default"},
},
"3": {
"class_type": "ConditioningLoader",
"inputs": {"conditioning_name": EMBEDDING_NAME},
},
"4": {
"class_type": "EmptyMiniMaxH3LatentAV",
"inputs": {"width": WIDTH, "height": HEIGHT, "length": frames},
},
"7": {"class_type": "RandomNoise", "inputs": {"noise_seed": seed}},
"8": {
"class_type": "KSamplerSelect",
"inputs": {"sampler_name": "res_multistep"},
},
"9": {
"class_type": "BasicScheduler",
"inputs": {
"model": None,
"scheduler": "simple",
"steps": STEPS,
"denoise": 1.0,
},
},
"10": {
"class_type": "BasicGuider",
"inputs": {"model": None, "conditioning": ["3", 0]},
},
"11": {
"class_type": "SamplerCustomAdvanced",
"inputs": {
"noise": ["7", 0],
"guider": ["10", 0],
"sampler": ["8", 0],
"sigmas": ["9", 0],
"latent_image": ["4", 0],
},
},
"12": {
"class_type": "SaveMiniMaxH3AVLatentCache",
"inputs": {"samples": ["11", 0], "cache_path": str(target)},
},
}
model = ["1", 0]
if method == "spark_10pct":
nodes["6"] = {
"class_type": "MiniMaxH3SparkAttentionSM120",
"inputs": {
"model": model,
"enabled": True,
"steps": STEPS,
"warmup_mode": "warmup_ratio",
"warmup_ratio": 0.2,
"warmup_steps": 4,
"topk_mode": "topk_ratio",
"topk_ratio": 0.1,
"topk_blocks": 114,
"dense_layers": 1,
"min_tokens": 12288,
"strict": True,
"tail_granularity": "query",
},
}
model = ["6", 0]
nodes["9"]["inputs"]["model"] = model
nodes["10"]["inputs"]["model"] = model
return nodes
def start_comfy_server():
for name in ("user", "input", "temp", "latents"):
(ROOT / "comfyui" / name).mkdir(parents=True, exist_ok=True)
log = (ROOT / "comfyui" / "server.log").open("a")
env = {
**os.environ,
"HF_HUB_OFFLINE": "1",
"PYTHONUNBUFFERED": "1",
"NO_PROXY": "127.0.0.1,localhost",
"no_proxy": "127.0.0.1,localhost",
}
command = [
str(PYTHON),
"main.py",
"--listen",
"127.0.0.1",
"--port",
str(PORT),
"--disable-auto-launch",
"--disable-cuda-malloc",
"--preview-method",
"none",
"--output-directory",
str(COMFY_OUT),
"--temp-directory",
str(ROOT / "comfyui" / "temp"),
"--input-directory",
str(ROOT / "comfyui" / "input"),
"--user-directory",
str(ROOT / "comfyui" / "user"),
]
return (
subprocess.Popen(command, cwd=COMFY, env=env, stdout=log, stderr=subprocess.STDOUT),
log,
)
def comfyui_worker() -> None:
sys.path.insert(0, str(REPO / "scripts"))
import comfyui_sol_4way_benchmark_20260923 as api
process, log = start_comfy_server()
try:
stats = api.wait_server(PORT)
write_json(ROOT / "comfyui" / "worker.json", stats)
for duration_index, (duration, shape) in enumerate(DURATIONS.items()):
method_order = METHODS[::-1] if duration_index % 2 == 0 else METHODS
for method in method_order:
for phase, repeat, seed in (
("warmup", 0, 41),
("measured", 1, MEASURED_SEEDS[0]),
("measured", 2, MEASURED_SEEDS[1]),
):
target = ROOT / "comfyui" / "latents" / (
f"{duration}_{method}_{phase}_{repeat}.safetensors"
)
result = api.queue_and_wait(
PORT,
comfy_graph(method, shape["model_frames"], seed, target),
"11",
)
if result.get("sampler_seconds") is None or not target.is_file():
raise RuntimeError(
f"missing ComfyUI timing or latent: {duration} {method} {phase}"
)
target.unlink()
result.pop("history", None)
record = {
"backend": "comfyui",
"duration": duration,
"method": method,
"phase": phase,
"repeat": repeat,
"seed": seed,
"model_frames": shape["model_frames"],
"seconds": result["sampler_seconds"],
**result,
}
write_json(
ROOT / "comfyui" / "records" / (
f"{duration}_{method}_{phase}_{repeat}.json"
),
record,
)
print(
"COMFYUI",
duration,
method,
phase,
repeat,
f"{record['seconds']:.3f}s",
flush=True,
)
finally:
process.terminate()
try:
process.wait(timeout=30)
except subprocess.TimeoutExpired:
process.kill()
process.wait()
log.close()
def load_measured(backend: str) -> list[dict]:
return [
json.loads(path.read_text())
for path in sorted((ROOT / backend / "records").glob("*.json"))
if json.loads(path.read_text())["phase"] == "measured"
]
def summarize() -> None:
summary = {}
for backend in ("diffusers", "comfyui"):
rows = load_measured(backend)
summary[backend] = {}
for duration in DURATIONS:
summary[backend][duration] = {}
for method in METHODS:
values = [
row["seconds"]
for row in rows
if row["duration"] == duration and row["method"] == method
]
if len(values) != len(MEASURED_SEEDS):
raise RuntimeError(
f"missing measurements: {backend} {duration} {method}: {values}"
)
summary[backend][duration][method] = {
"seconds": values,
"median_seconds": statistics.median(values),
"mean_seconds": statistics.mean(values),
"range_seconds": max(values) - min(values),
}
dense = summary[backend][duration]["dense"]["median_seconds"]
spark = summary[backend][duration]["spark_10pct"]["median_seconds"]
summary[backend][duration]["speedup_x"] = dense / spark
result = {
"status": "complete",
"completed_unix": time.time(),
"summary": summary,
}
write_json(ROOT / "results.json", result)
protocol = json.loads((ROOT / "protocol.json").read_text())
protocol["status"] = "complete"
protocol["completed_unix"] = result["completed_unix"]
write_json(ROOT / "protocol.json", protocol)
lines = [
"# Current warmup: Diffusers vs ComfyUI Dense/Spark-10",
"",
"One excluded full generation per backend/method/shape; two measured generations. "
"Speedup is the ratio of Dense median to Spark median.",
"",
"| Backend | Duration | Dense runs (s) | Spark-10 runs (s) | Dense median (s) | Spark median (s) | Speedup |",
"|---|---:|---|---|---:|---:|---:|",
]
for backend in ("diffusers", "comfyui"):
for duration in DURATIONS:
row = summary[backend][duration]
dense = row["dense"]
spark = row["spark_10pct"]
lines.append(
f"| {backend} | {duration.replace('14p4s', '14.4s')} | "
f"{', '.join(f'{x:.2f}' for x in dense['seconds'])} | "
f"{', '.join(f'{x:.2f}' for x in spark['seconds'])} | "
f"{dense['median_seconds']:.2f} | {spark['median_seconds']:.2f} | "
f"{row['speedup_x']:.3f}x |"
)
lines.extend(
[
"",
"Diffusers requests 120/240/345 frames and internally aligns them to "
"124/243/345; ComfyUI receives the aligned lengths directly. Diffusers "
"executes 19 transformer evaluations for 20 requested scheduler points, "
"whereas ComfyUI executes 20 sampler model evaluations. Both use four "
"initial dense evaluations and dense transformer layer 0.",
]
)
(ROOT / "REPORT.md").write_text("\n".join(lines) + "\n")
print(json.dumps(result, indent=2), flush=True)
def run_all() -> None:
prepare()
common = {
**os.environ,
"HF_HUB_OFFLINE": "1",
"OMP_NUM_THREADS": "4",
"TORCHINDUCTOR_COMPILE_THREADS": "4",
"PYTHONUNBUFFERED": "1",
"PYTHONDONTWRITEBYTECODE": "1",
"FLASHINFER_CUDA_ARCH_LIST": "12.0",
"H3_SOL_LAYOUT_FAST": "1",
"H3_METRIC_FACTOR": "cholesky",
"H3_LMV2_COS_PRECISION": "fp16",
"H3_LMV2_FUSED_NODE": "1",
"H3_LMV2_GROUP1_FAST": "1",
"H3_LMV2_COS_FAST": "1",
"H3_LMV2_SMALL_PROXY_FAST": "1",
"H3_LMV2_FP8_FEATURES": "0",
"H3_IMPL_REPO": str(REPO),
}
jobs = []
for backend, gpu in (("diffusers", 0), ("comfyui", 1)):
log = (ROOT / f"{backend}.log").open("a")
env = {**common, "CUDA_VISIBLE_DEVICES": str(gpu)}
process = subprocess.Popen(
[str(PYTHON), str(Path(__file__).resolve()), f"{backend}-worker"],
cwd=REPO,
env=env,
stdout=log,
stderr=subprocess.STDOUT,
)
jobs.append((backend, process, log))
failure = None
while any(process.poll() is None for _, process, _ in jobs):
for backend, process, _ in jobs:
if process.poll() not in (None, 0):
failure = f"{backend} worker failed with exit code {process.returncode}"
break
if failure:
break
time.sleep(5)
if failure:
for _, process, _ in jobs:
if process.poll() is None:
process.terminate()
for _, process, log in jobs:
process.wait()
log.close()
write_json(ROOT / "failure.json", {"status": "failed", "error": failure})
raise RuntimeError(failure)
for backend, process, log in jobs:
returncode = process.wait()
log.close()
if returncode:
raise RuntimeError(f"{backend} worker failed with exit code {returncode}")
summarize()
if __name__ == "__main__":
command = sys.argv[1] if len(sys.argv) > 1 else "run"
if command == "run":
run_all()
elif command == "diffusers-worker":
diffusers_worker()
elif command == "comfyui-worker":
comfyui_worker()
elif command == "summarize":
summarize()
else:
raise ValueError(command)