# SPDX-License-Identifier: Apache-2.0 """Viggle-Animate: replace the performers in a video with the person in a still. python inference/sample.py --cond driving.mp4 --ref character.png --out swapped.mp4 `--cond` supplies the motion, camera framing, background and lighting; `--ref` supplies who is in it. Everything about the render except the people is copied from `--cond`. By default the model generates its own soundtrack, and the fixed prompt asks for silence -- so there is nothing for the mouth to sync to. `--audio` pins a real one instead: python inference/sample.py --cond driving.mp4 --ref character.png \ --audio driving.mp4 --out swapped.mp4 The soundtrack is encoded once and held in the target audio rows as a *clean* latent for the whole denoise, so the model conditions on it rather than predicting it, and the mouth tracks that speech. The track written to `--out` is then that same audio, back through the audio VAE. The text encoder is never loaded. Conditioning comes from `assets/fixed_embed_fwd_anyframe.pt`, a frozen 362 x 5120 tensor computed once from the fixed prompt in `assets/fixed_prompt.txt`, so Qwen3-VL (63 GB of the base repo) stays on disk and the text block of the packed sequence is 362 rows instead of several thousand. There is no per-clip prompt and no caption: nothing in the output comes from text you write. Needs `--model-dir` pointing at a local copy of MiniMaxAI/MiniMax-H3 for the VAE, the audio VAE and the schedulers. This repository ships only the transformer and the LoRA. """ import argparse import os import time import torch from diffusers import MiniMaxH3Transformer3DModel, ModularPipeline from diffusers.modular_pipelines.minimax_h3 import (MiniMaxH3AudioReference, MiniMaxH3ImageReference, MiniMaxH3VideoReference) from diffusers.modular_pipelines.minimax_h3.before_encoder import MiniMaxH3Ref2VASetupStep from diffusers.modular_pipelines.minimax_h3.encoders import MiniMaxH3Ref2VATextEncoderStep from diffusers.modular_pipelines.minimax_h3.modular_pipeline import (align_num_frames, audio_latent_num_frames) from diffusers.utils.export_utils import encode_video HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) parser = argparse.ArgumentParser() parser.add_argument("--cond", required=True, help="the video whose motion, framing and background are kept") parser.add_argument("--ref", required=True, help="a single still of the person to put in it") parser.add_argument("--out", required=True) parser.add_argument("--model-dir", required=True, help="a local copy of MiniMaxAI/MiniMax-H3, for the VAE / audio VAE / schedulers") parser.add_argument("--transformer", default=os.path.join(HERE, "transformer")) parser.add_argument("--lora", default=os.path.join(HERE, "lora")) parser.add_argument("--embed", default=os.path.join(HERE, "assets", "fixed_embed_fwd_anyframe.pt")) parser.add_argument("--num-frames", type=int, default=124, help="at 24 fps; 124 frames is ~5.2 s") parser.add_argument("--steps", type=int, default=4, help="the distilled student's operating point. More is not monotonically better: " "s4 is not a degraded s12") parser.add_argument("--flow-shift", type=float, default=3.0, help="the base model's released default is 12; the few-step student wants 3") parser.add_argument("--height", type=int, default=None, help="defaults to the conditioning clip's own height") parser.add_argument("--width", type=int, default=None, help="defaults to the conditioning clip's own width") parser.add_argument("--short-edge", type=int, default=None, help="the canvas both references are laid out on. Defaults to the conditioning clip's own " "short edge, which is what this model was evaluated at") parser.add_argument("--offload", action="store_true", help="stream the transformer from CPU in groups of 5 blocks: ~12 GB resident instead of 62") parser.add_argument("--audio", default=None, help="pin the generated soundtrack to this file's audio -- usually the driving clip " "itself -- so the mouth tracks real speech instead of the silence the fixed " "prompt asks for. Any file PyAV can decode; a video's soundtrack is taken") parser.add_argument("--seed", type=int, default=42) args = parser.parse_args() fixed = torch.load(args.embed, weights_only=False) def use_fixed_embeds(self, components, state): block_state = self.get_block_state(state) block_state.prompt_embeds = fixed["prompt_embeds"].to(components._execution_device, torch.bfloat16) block_state.text_token_tags = fixed["text_token_tags"] self.set_block_state(state, block_state) return components, state MiniMaxH3Ref2VATextEncoderStep.__call__ = use_fixed_embeds # `--audio` holds the *target* audio rows at a real soundtrack, clean, for the whole denoise, rather # than letting the model generate them. Two of the three things that takes have no argument on the # pipeline, so they are patched here; the third is the `audio_latents=` passed to the call below. if args.audio: from diffusers.modular_pipelines.minimax_h3.before_denoise import MiniMaxH3SetTimestepsStep from diffusers.modular_pipelines.minimax_h3.denoise import MiniMaxH3LoopSchedulerStep # (1) Those rows carry finished audio, so they have to be told they are clean. H3's flow # convention is reversed -- t = 1 is clean, not 0 -- and the library itself passes a literal 1.0 # for a *reference* soundtrack. `audio_timestep` is positional argument 6. _build_row_timesteps = MiniMaxH3SetTimestepsStep.build_row_timesteps MiniMaxH3SetTimestepsStep.build_row_timesteps = staticmethod( lambda *a: _build_row_timesteps(*a[:6], 1.0, *a[7:])) # (2) ...and the scheduler must never write them, or the first step would walk them off the # soundtrack. Only the video rows are stepped. `num_condition_audio_rows` deliberately stays 0: # raising it empties the decoder's `audio_latents[num_condition_audio_rows:]` slice and trips the # reference-count check, and these rows are a pinned target, not a reference. @torch.no_grad() def video_only_step(self, components, block_state, i, t): n = block_state.num_condition_video_rows block_state.latents[n:] = components.scheduler.step( block_state.noise_pred[0, n:].float(), t, block_state.latents[n:], return_dict=False)[0] return components, block_state MiniMaxH3LoopSchedulerStep.__call__ = video_only_step # The reference order is frozen: the presentation names `