kelseye commited on
Commit
92c8dff
·
verified ·
1 Parent(s): ff8de48

Upload folder using huggingface_hub

Browse files
.gitattributes CHANGED
@@ -1,35 +1,60 @@
1
  *.7z filter=lfs diff=lfs merge=lfs -text
2
  *.arrow filter=lfs diff=lfs merge=lfs -text
3
  *.bin filter=lfs diff=lfs merge=lfs -text
 
4
  *.bz2 filter=lfs diff=lfs merge=lfs -text
5
- *.ckpt filter=lfs diff=lfs merge=lfs -text
6
  *.ftz filter=lfs diff=lfs merge=lfs -text
7
  *.gz filter=lfs diff=lfs merge=lfs -text
8
  *.h5 filter=lfs diff=lfs merge=lfs -text
9
  *.joblib filter=lfs diff=lfs merge=lfs -text
10
  *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
- *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
  *.model filter=lfs diff=lfs merge=lfs -text
13
  *.msgpack filter=lfs diff=lfs merge=lfs -text
14
- *.npy filter=lfs diff=lfs merge=lfs -text
15
- *.npz filter=lfs diff=lfs merge=lfs -text
16
  *.onnx filter=lfs diff=lfs merge=lfs -text
17
  *.ot filter=lfs diff=lfs merge=lfs -text
18
  *.parquet filter=lfs diff=lfs merge=lfs -text
19
  *.pb filter=lfs diff=lfs merge=lfs -text
20
- *.pickle filter=lfs diff=lfs merge=lfs -text
21
- *.pkl filter=lfs diff=lfs merge=lfs -text
22
  *.pt filter=lfs diff=lfs merge=lfs -text
23
  *.pth filter=lfs diff=lfs merge=lfs -text
24
  *.rar filter=lfs diff=lfs merge=lfs -text
25
- *.safetensors filter=lfs diff=lfs merge=lfs -text
26
  saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
  *.tar.* filter=lfs diff=lfs merge=lfs -text
28
- *.tar filter=lfs diff=lfs merge=lfs -text
29
  *.tflite filter=lfs diff=lfs merge=lfs -text
30
  *.tgz filter=lfs diff=lfs merge=lfs -text
31
- *.wasm filter=lfs diff=lfs merge=lfs -text
32
  *.xz filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  *.7z filter=lfs diff=lfs merge=lfs -text
2
  *.arrow filter=lfs diff=lfs merge=lfs -text
3
  *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bin.* filter=lfs diff=lfs merge=lfs -text
5
  *.bz2 filter=lfs diff=lfs merge=lfs -text
 
6
  *.ftz filter=lfs diff=lfs merge=lfs -text
7
  *.gz filter=lfs diff=lfs merge=lfs -text
8
  *.h5 filter=lfs diff=lfs merge=lfs -text
9
  *.joblib filter=lfs diff=lfs merge=lfs -text
10
  *.lfs.* filter=lfs diff=lfs merge=lfs -text
 
11
  *.model filter=lfs diff=lfs merge=lfs -text
12
  *.msgpack filter=lfs diff=lfs merge=lfs -text
 
 
13
  *.onnx filter=lfs diff=lfs merge=lfs -text
14
  *.ot filter=lfs diff=lfs merge=lfs -text
15
  *.parquet filter=lfs diff=lfs merge=lfs -text
16
  *.pb filter=lfs diff=lfs merge=lfs -text
 
 
17
  *.pt filter=lfs diff=lfs merge=lfs -text
18
  *.pth filter=lfs diff=lfs merge=lfs -text
19
  *.rar filter=lfs diff=lfs merge=lfs -text
 
20
  saved_model/**/* filter=lfs diff=lfs merge=lfs -text
21
  *.tar.* filter=lfs diff=lfs merge=lfs -text
 
22
  *.tflite filter=lfs diff=lfs merge=lfs -text
23
  *.tgz filter=lfs diff=lfs merge=lfs -text
 
24
  *.xz filter=lfs diff=lfs merge=lfs -text
25
  *.zip filter=lfs diff=lfs merge=lfs -text
26
+ *.zstandard filter=lfs diff=lfs merge=lfs -text
27
+ *.tfevents* filter=lfs diff=lfs merge=lfs -text
28
+ *.db* filter=lfs diff=lfs merge=lfs -text
29
+ *.ark* filter=lfs diff=lfs merge=lfs -text
30
+ **/*ckpt*data* filter=lfs diff=lfs merge=lfs -text
31
+ **/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text
32
+ **/*ckpt*.index filter=lfs diff=lfs merge=lfs -text
33
+
34
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
35
+ *.gguf* filter=lfs diff=lfs merge=lfs -text
36
+ *.ggml filter=lfs diff=lfs merge=lfs -text
37
+ *.llamafile* filter=lfs diff=lfs merge=lfs -text
38
+ *.pt2 filter=lfs diff=lfs merge=lfs -text
39
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
40
+ *.npy filter=lfs diff=lfs merge=lfs -text
41
+ *.npz filter=lfs diff=lfs merge=lfs -text
42
+ *.pickle filter=lfs diff=lfs merge=lfs -text
43
+ *.pkl filter=lfs diff=lfs merge=lfs -text
44
+ *.tar filter=lfs diff=lfs merge=lfs -text
45
+ *.wasm filter=lfs diff=lfs merge=lfs -text
46
  *.zst filter=lfs diff=lfs merge=lfs -text
47
  *tfevents* filter=lfs diff=lfs merge=lfs -text
48
+
49
+ ./minimax-h3-ref2va-nf4.safetensors filter=lfs diff=lfs merge=lfs -text
50
+ ./minimax-h3-text-encoder-nf4.safetensors filter=lfs diff=lfs merge=lfs -text
51
+ ./minimax-h3-fl2va-nf4.safetensors filter=lfs diff=lfs merge=lfs -text
52
+ "/video_vae_nf4.safetensors" filter=lfs diff=lfs merge=lfs -text
53
+ "/audio_vae_nf4.safetensors" filter=lfs diff=lfs merge=lfs -text
54
+ "/minimax-h3-fl2va-nf4.safetensors" filter=lfs diff=lfs merge=lfs -text
55
+ "/minimax-h3-ref2va-nf4.safetensors" filter=lfs diff=lfs merge=lfs -text
56
+ "/minimax-h3-text-encoder-nf4.safetensors" filter=lfs diff=lfs merge=lfs -textaudio_vae_nf4.safetensors filter=lfs diff=lfs merge=lfs -text
57
+ minimax-h3-fl2va-nf4.safetensors filter=lfs diff=lfs merge=lfs -text
58
+ minimax-h3-ref2va-nf4.safetensors filter=lfs diff=lfs merge=lfs -text
59
+ minimax-h3-text-encoder-nf4.safetensors filter=lfs diff=lfs merge=lfs -text
60
+ video_vae_nf4.safetensors filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,212 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ ---
4
+ # MiniMax-H3-NF4
5
+
6
+ The **NF4 quantized version** of the MiniMax-H3 multimodal audio-video generation model (4-bit quantized via `bitsandbytes`), designed for use with [DiffSynth-Studio](https://github.com/modelscope/DiffSynth-Studio). Enables joint "text/image/video/audio → video + audio" generation on machines with limited GPU memory or system RAM.
7
+
8
+ ## File Overview
9
+
10
+ | File | Size | Function | Shared? |
11
+ |---|---|---|---|
12
+ | `minimax-h3-fl2va-nf4.safetensors` | ~16 GB | DiT backbone for **FL2VA** task (text / start-end keyframes → video+audio) | FL2VA only |
13
+ | `minimax-h3-ref2va-nf4.safetensors` | ~16 GB | DiT backbone for **Ref2VA** task (reference image/video/audio → video+audio) | Ref2VA only |
14
+ | `minimax-h3-text-encoder-nf4.safetensors` | ~15 GB | Qwen3-VL text/visual encoder | Shared across tasks |
15
+ | `video_vae_nf4.safetensors` | ~1.6 GB | Video VAE decoder | Shared across tasks |
16
+ | `audio_vae_nf4.safetensors` | ~271 MB | Audio VAE decoder | Shared across tasks |
17
+
18
+ > Note: Choose one DiT model depending on the task; the other three components (text_encoder / video_vae / audio_vae) are shared between both tasks. The framework automatically identifies component types and applies appropriate quantization configurations (including bf16 fallback for a few quantization-sensitive layers) based on file hashes—no manual configuration required.
19
+
20
+ ## System Requirements
21
+
22
+ - CUDA GPU (NF4 dequantization relies on `bitsandbytes` CUDA kernels)
23
+ - Processor and tokenizer must be obtained from the original repository `MiniMax/MiniMax-H3` (see `processor_config` below)
24
+
25
+ ### Install DiffSynth-Studio
26
+
27
+ Install from source (recommended, ensures latest MiniMax-H3 support), including NF4 quantization dependencies:
28
+
29
+ ```bash
30
+ git clone https://github.com/modelscope/DiffSynth-Studio.git
31
+ cd DiffSynth-Studio
32
+ pip install -e ".[quant]"
33
+ ```
34
+
35
+ ## Usage (Disk Offload, Low VRAM & RAM)
36
+
37
+ Weights remain on disk and are streamed into GPU layer-by-layer during inference, minimizing VRAM usage. **Text-to-video+audio (t2v) can run with as little as ~6 GB VRAM.**
38
+
39
+ > `vram_limit` (in GB) sets the VRAM threshold—lower values reduce memory usage at the cost of speed.
40
+
41
+ > If you have sufficient CPU RAM, set `offload_device` / `offload_dtype` to `"cpu"` / `torch.bfloat16` (i.e., CPU offload). This keeps weights in main memory instead of reading from disk, resulting in faster performance; all other code remains unchanged.
42
+
43
+ ### FL2VA — Text / Start-End Keyframes → Video + Audio
44
+
45
+ ```python
46
+ import torch
47
+ from diffsynth.pipelines.minimax_h3_audio_video import MiniMaxH3Pipeline, ModelConfig
48
+ from diffsynth.utils.data.audio_video import write_video_audio
49
+ from modelscope import dataset_snapshot_download
50
+ from PIL import Image
51
+ ```
52
+
53
+ vram_config = {
54
+ "offload_dtype": "disk",
55
+ "offload_device": "disk",
56
+ "onload_dtype": torch.bfloat16,
57
+ "onload_device": "cpu",
58
+ "preparing_dtype": torch.bfloat16,
59
+ "preparing_device": "cuda",
60
+ "computation_dtype": torch.bfloat16,
61
+ "computation_device": "cuda",
62
+ }
63
+ pipe = MiniMaxH3Pipeline.from_pretrained(
64
+ torch_dtype=torch.bfloat16,
65
+ device="cuda",
66
+ model_configs=[
67
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-fl2va-nf4.safetensors", **vram_config),
68
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-text-encoder-nf4.safetensors", **vram_config),
69
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="video_vae_nf4.safetensors", **vram_config),
70
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="audio_vae_nf4.safetensors", **vram_config),
71
+ ],
72
+ processor_config=ModelConfig(model_id="MiniMax/MiniMax-H3", origin_file_pattern="FL2VA/processor/"),
73
+ vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 2,
74
+ )
75
+
76
+ # Text -> Video + Audio
77
+ prompt = "A girl is very happy, she is speaking in english: “I enjoy working with Diffsynth-Studio, it's a perfect framework.”"
78
+ video, audio = pipe(
79
+ prompt=prompt,
80
+ height=480, width=832, num_frames=124, num_inference_steps=50, seed=0,
81
+ )
82
+ write_video_audio(
83
+ video=video, audio=audio,
84
+ output_path="t2va.mp4", fps=24, audio_sample_rate=32000,
85
+ )
86
+
87
+ # Text + First Frame + Last Frame -> Video + Audio
88
+ dataset_snapshot_download(dataset_id="DiffSynth-Studio/diffsynth_example_dataset", local_dir="data/diffsynth_example_dataset", allow_file_pattern="minimax_h3/MiniMax-H3-FL2VA/*")
89
+ first_frame = Image.open("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-FL2VA/first.png")
90
+ last_frame = Image.open("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-FL2VA/last.png")
91
+ prompt = "A short indoor drama scene of a family argument, vertical video format with short-form video aesthetics, realistic live-action performance, Chinese household or small restaurant interior setting, warm lighting, red decorations and calligraphy scrolls in the background, shallow depth of field, intense emotions, fast-paced editing. Performance requirements: authentic short-video acting style, no exaggerated theatrical tone. The man speaks with anger, grievance, and urgent rebuttal, saying 'What exactly do you want?' The middle-aged woman speaks sharply, assertively, and aggressively demanding, saying 'You must pay up!' There should be strong confrontation between them, escalating in intensity. Visual style: vertical 9:16 aspect ratio, smartphone short-video look, realistic live-action footage, shallow depth of field, warm indoor lighting, mostly medium and close-up shots, frequent shot-reverse-shot editing, background should remain everyday and realistic—no sci-fi, no historical costumes, no animation-like visuals. No subtitles, text, platform watermarks, or overlays should appear in the画面."
92
+ video, audio = pipe(
93
+ prompt=prompt,
94
+ height=832, width=480, num_frames=124, num_inference_steps=50, seed=0,
95
+ keyframes=[first_frame, last_frame], keyframe_indices=[0, -1],
96
+ )
97
+ write_video_audio(
98
+ video=video, audio=audio,
99
+ output_path="fl2va.mp4", fps=24, audio_sample_rate=32000,
100
+ )
101
+ ```
102
+
103
+ ### Ref2VA — Reference Image/Video/Audio → Video + Audio
104
+
105
+ Four types of references are supported, which can be combined within a single list (`video` is silent; for videos with sound, use `video_audio`):
106
+
107
+ ```python
108
+ {"type": "image", "image": PIL.Image}
109
+ {"type": "video", "video": list[PIL.Image]} # silent
110
+ {"type": "audio", "audio": Tensor[C, L], "sample_rate": int}
111
+ {"type": "video_audio", "video": list[PIL.Image], "audio": Tensor[C, L], "sample_rate": int}
112
+ ```
113
+
114
+ ```python
115
+ import torch
116
+ from PIL import Image
117
+ from diffsynth.pipelines.minimax_h3_audio_video import MiniMaxH3Pipeline, ModelConfig
118
+ from diffsynth.utils.data.audio_video import write_video_audio
119
+ from diffsynth.utils.data.audio import read_audio
120
+ from diffsynth.utils.data import VideoData
121
+ from modelscope import dataset_snapshot_download
122
+
123
+ def align_frame_count(frame_count):
124
+ current = max(int(frame_count), 1)
125
+ while current % 17 != 5:
126
+ current += 1
127
+ return current
128
+
129
+ def read_video_with_fps(path, num_out_frames, height, width, fps=24):
130
+ video = VideoData(path, height=height, width=width)
131
+ frames = video.raw_data()
132
+ src_fps = float(video.data.reader.get_meta_data()["fps"])
133
+ out = []
134
+ for k in range(num_out_frames):
135
+ idx = int(round(k * src_fps / fps))
136
+ if idx >= len(frames):
137
+ break
138
+ out.append(frames[idx])
139
+ return out
140
+
141
+ vram_config = {
142
+ "offload_dtype": "disk",
143
+ "offload_device": "disk",
144
+ "onload_dtype": torch.bfloat16,
145
+ "onload_device": "cpu",
146
+ "preparing_dtype": torch.bfloat16,
147
+ "preparing_device": "cuda",
148
+ "computation_dtype": torch.bfloat16,
149
+ "computation_device": "cuda",
150
+ }
151
+ pipe = MiniMaxH3Pipeline.from_pretrained(
152
+ torch_dtype=torch.bfloat16,
153
+ device="cuda",
154
+ model_configs=[
155
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-ref2va-nf4.safetensors", **vram_config),
156
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-text-encoder-nf4.safetensors", **vram_config),
157
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="video_vae_nf4.safetensors", **vram_config),
158
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="audio_vae_nf4.safetensors", **vram_config),
159
+ ],
160
+ processor_config=ModelConfig(model_id="MiniMax/MiniMax-H3", origin_file_pattern="Ref2VA/processor/"),
161
+ vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 5,
162
+ )
163
+
164
+ # Text + Reference Image -> Video + Audio
165
+ dataset_snapshot_download(dataset_id="DiffSynth-Studio/diffsynth_example_dataset", local_dir="data/diffsynth_example_dataset", allow_file_pattern="minimax_h3/MiniMax-H3-Ref2VA/*")
166
+ ref_image = Image.open("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/0.png").convert("RGB")
167
+ prompt = "A website page, UI design of a web page, web animation. The video demonstrates a smooth scrolling effect downward. A highly dynamic and energetic product official website-style landing page UI/UX demonstration video, with the main focus being product image 1. The layout features bold, slanted, oversized sans-serif typography. The background includes fast-paced dynamic lighting effects intertwined with dark carbon fiber or sporty breathable mesh textures in motion. The video showcases a tightly paced, powerful downward scroll effect, along with strong visual interactions such as significant zoom-in and color inversion when hovering over UI elements."
168
+ video, audio = pipe(
169
+ prompt=prompt,
170
+ height=480, width=832, num_frames=124, num_inference_steps=50, seed=42,
171
+ references=[{"type": "image", "image": Image.open("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/0.png").convert("RGB")}]
172
+ )
173
+ write_video_audio(
174
+ video=video, audio=audio,
175
+ output_path="ti2va.mp4", fps=24, audio_sample_rate=32000,
176
+ )
177
+
178
+ # Text + Reference Audio + Reference Video -> Video + Audio
179
+ ref_video = read_video_with_fps("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/video.mp4", 124, 480, 832)
180
+ ref_audio, sample_rate = read_audio("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/voice.mp3", duration=len(ref_video) / 24, resample=True, resample_rate=pipe.audio_vae.sample_rate)
181
+ prompt = "subject_definitions:\n<Subject 1> is the young man with short wavy blonde hair, wearing a bright pink suit jacket, matching pink trousers, an unbuttoned white shirt, and silver rings, holding a small black lamb in his arms in <Video 1>.\n<Video 1> is the source video for the editing task.\n<Audio 1> is the synchronized audio track of <Video 1>, providing the background music.\n<Audio 2> is the voice timbre reference for <Subject 1>'s voice, containing a spoken male voiceover.\n\nsummary:\n[video editing + audio reference + audio reuse] The target video is an edited version of <Video 1>. <Subject 1>, wearing a bright pink suit and holding a black lamb, stands in a grassy field with other white lambs in the background. The edit animates <Subject 1>'s face to speak the user-provided dialogue. <Audio 1> is partially reused as the continuous background music, while the target references the calm male voice timbre of <Audio 2> for <Subject 1>'s spoken lines.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1]): fully_preserved - the man retains his identity, wavy blonde hair, pink suit, white shirt, accessories, and the black lamb he holds, with his mouth newly animated to speak.\n<Video 1> (source video editing): fully_preserved - the original camera framing, warm golden hour lighting, grassy hill setting, and background white lambs are maintained while the central character is edited.\n<Audio 1>: partially_copy - the atmospheric background music from <Audio 1> is reused in the target video, mixed beneath the newly added spoken dialogue.\n<Audio 2>: reference - the target audio references the male voice timbre from <Audio 2> to generate <Subject 1>'s spoken dialogue.\n\ndetailed_description:\nThe target video is in realistic photographic style.\n[Shot 1] The shot begins from the source <Video 1>, showing <Subject 1>, a young man with short wavy blonde hair, wearing a bright pink suit jacket, matching pink trousers, and a casually unbuttoned white shirt. He stands confidently in a sunlit green pasture, gently holding a small black lamb securely in his arms. The warm, golden hour lighting casts soft shadows across his face and the bright pink fabric of his suit. Behind him, several white lambs stand and graze on the rolling grassy hill against a clear, pale blue sky. The atmospheric background music from <Audio 1> plays continuously throughout the scene. <Subject 1> physically speaks, his mouth movements naturally syncing to the new dialogue, with his voice timbre referencing the calm male delivery from <Audio 2>. Looking thoughtfully forward, <Subject 1> (S1) speaks softly, <d>[English] Follow the wind, live free.</d> As he delivers the line, he subtly shifts his weight, cradling the resting black lamb while the camera slowly pushes in. <Subject 1> (S1) continues his thought, <d>[English] Leave worries behind, enjoy the moment.</d> Exactly as his voice stops, his lips meet in a relaxed, peaceful smile, and his jaw ceases speaking motion. He then turns his gaze slightly away toward the horizon, gently stroking the black lamb's fleece with his fingers as the camera holds on this tranquil, sunlit state through the end of the video.\n\noverall_soundscape:\nThe soundscape consists of the continuous, atmospheric background music from <Audio 1>, overlaid with the clear, calm male dialogue spoken by the main character, referencing the voice timbre of <Audio 2>.\n\nnon_diegetic_music:\nThe atmospheric, sustained background music from <Audio 1> is reused as the continuous score, playing quietly beneath the spoken dialogue."
182
+ video, audio = pipe(
183
+ prompt=prompt,
184
+ height=480, width=832, num_frames=124, num_inference_steps=50, seed=42,
185
+ references=[
186
+ {"type": "video", "video": ref_video},
187
+ {"type": "audio", "audio": ref_audio, "sample_rate": sample_rate},
188
+ ],
189
+ )
190
+ write_video_audio(
191
+ video=video, audio=audio,
192
+ output_path="tav2va.mp4", fps=24, audio_sample_rate=32000,
193
+ )
194
+ ```
195
+
196
+ ## Common Parameters
197
+
198
+ - `height` / `width`: Resolution, e.g., `480x832` (landscape) or `832x480` (portrait).
199
+ - `num_frames`: Number of frames, must satisfy `num_frames % 17 == 5` (e.g., 124).
200
+ - `num_inference_steps`: Denoising steps, example uses 50.
201
+ - `keyframes` / `keyframe_indices`: FL2VA control for start and end frames (`[0, -1]` means first and last frame).
202
+ - `references`: Ref2VA reference list, elements are `{"type": "image|video|audio|video_audio", ...}`.
203
+ - Output: `write_video_audio(video, audio, output_path, fps=24, audio_sample_rate=32000)`.
204
+
205
+ ## Example Scripts
206
+
207
+ Fully runnable scripts in the repository (examples in this README are derived from these):
208
+
209
+ - `examples/minimax_h3/model_inference_low_vram/MiniMax-H3-NF4-FL2VA.py`
210
+ - `examples/minimax_h3/model_inference_low_vram/MiniMax-H3-NF4-Ref2VA.py`
211
+ - `examples/minimax_h3/model_inference/MiniMax-H3-NF4-FL2VA.py` (CPU offload)
212
+ - `examples/minimax_h3/model_inference/MiniMax-H3-NF4-Ref2VA.py` (CPU offload)
README_from_modelscope.md ADDED
@@ -0,0 +1,216 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model:
3
+ - MiniMax/MiniMax-H3
4
+ frameworks:
5
+ - ""
6
+ license: Apache License 2.0
7
+ base_model_relation: quantized
8
+ ---
9
+ # MiniMax-H3-NF4
10
+
11
+ MiniMax-H3 多模态音视频生成模型的 **NF4 量化版本**(通过 `bitsandbytes` 4-bit 量化),配合 [DiffSynth-Studio](https://github.com/modelscope/DiffSynth-Studio) 使用,可在显存/内存受限的机器上进行「文本/图像/视频/音频 → 视频 + 音频」的联合生成。
12
+
13
+ ## 文件说明
14
+
15
+ | 文件 | 大小 | 作用 | 是否共用 |
16
+ |---|---|---|---|
17
+ | `minimax-h3-fl2va-nf4.safetensors` | ~16 GB | **FL2VA** 任务的 DiT 主干(文本 / 首尾关键帧 → 视频+音频) | FL2VA 专用 |
18
+ | `minimax-h3-ref2va-nf4.safetensors` | ~16 GB | **Ref2VA** 任务的 DiT 主干(参考图像/视频/音频 → 视频+音频) | Ref2VA 专用 |
19
+ | `minimax-h3-text-encoder-nf4.safetensors` | ~15 GB | Qwen3-VL 文本/视觉编码器 | 两任务共用 |
20
+ | `video_vae_nf4.safetensors` | ~1.6 GB | 视频 VAE 解码器 | 两任务共用 |
21
+ | `audio_vae_nf4.safetensors` | ~271 MB | 音频 VAE 解码器 | 两任务共用 |
22
+
23
+ > 说明:DiT 按任务二选一,其余三个(text_encoder / video_vae / audio_vae)在两种任务下通用。加载时框架会根据文件 hash 自动识别组件类型并套用对应的量化配置(含对少数量化敏感层的 bf16 保留),无需手动指定量化参数。
24
+
25
+ ## 环境要求
26
+
27
+ - CUDA GPU(NF4 反量化依赖 `bitsandbytes` 的 CUDA kernel)
28
+ - processor / tokenizer 需从原始仓库 `MiniMax/MiniMax-H3` 获取(下面 `processor_config`)
29
+
30
+ ### 安装 DiffSynth-Studio
31
+
32
+ 从源码安装(推荐,可获得最新的 MiniMax-H3 支持),并直接带上 NF4 量化依赖:
33
+
34
+ ```bash
35
+ git clone https://github.com/modelscope/DiffSynth-Studio.git
36
+ cd DiffSynth-Studio
37
+ pip install -e ".[quant]"
38
+ ```
39
+
40
+ ## 使用(Disk offload,低显存和内存占用)
41
+
42
+ 权重存放磁盘、推理时按层流式加载到 GPU,显存占用最低。**纯文本生成视频+音频(t2v)最低约 6 GB 显存即可运行。**
43
+
44
+ > `vram_limit`(单位 GB)是显存占用阈值,调小可降低显存占用(代价是更慢)。
45
+
46
+ > 若 CPU 内存充足,可把 `offload_device` / `offload_dtype` 改为 `"cpu"` / `torch.bfloat16`(即 CPU offload),权重常驻内存、不走磁盘,速度更快;其余代码不变。
47
+
48
+ ### FL2VA — 文本 / 首尾关键帧 → 视频+音频
49
+
50
+ ```python
51
+ import torch
52
+ from diffsynth.pipelines.minimax_h3_audio_video import MiniMaxH3Pipeline, ModelConfig
53
+ from diffsynth.utils.data.audio_video import write_video_audio
54
+ from modelscope import dataset_snapshot_download
55
+ from PIL import Image
56
+
57
+ vram_config = {
58
+ "offload_dtype": "disk",
59
+ "offload_device": "disk",
60
+ "onload_dtype": torch.bfloat16,
61
+ "onload_device": "cpu",
62
+ "preparing_dtype": torch.bfloat16,
63
+ "preparing_device": "cuda",
64
+ "computation_dtype": torch.bfloat16,
65
+ "computation_device": "cuda",
66
+ }
67
+ pipe = MiniMaxH3Pipeline.from_pretrained(
68
+ torch_dtype=torch.bfloat16,
69
+ device="cuda",
70
+ model_configs=[
71
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-fl2va-nf4.safetensors", **vram_config),
72
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-text-encoder-nf4.safetensors", **vram_config),
73
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="video_vae_nf4.safetensors", **vram_config),
74
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="audio_vae_nf4.safetensors", **vram_config),
75
+ ],
76
+ processor_config=ModelConfig(model_id="MiniMax/MiniMax-H3", origin_file_pattern="FL2VA/processor/"),
77
+ vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 2,
78
+ )
79
+
80
+ # Text -> Video + Audio
81
+ prompt = "A girl is very happy, she is speaking in english: “I enjoy working with Diffsynth-Studio, it's a perfect framework.”"
82
+ video, audio = pipe(
83
+ prompt=prompt,
84
+ height=480, width=832, num_frames=124, num_inference_steps=50, seed=0,
85
+ )
86
+ write_video_audio(
87
+ video=video, audio=audio,
88
+ output_path="t2va.mp4", fps=24, audio_sample_rate=32000,
89
+ )
90
+
91
+ # Text + First Frame + Last Frame -> Video + Audio
92
+ dataset_snapshot_download(dataset_id="DiffSynth-Studio/diffsynth_example_dataset", local_dir="data/diffsynth_example_dataset", allow_file_pattern="minimax_h3/MiniMax-H3-FL2VA/*")
93
+ first_frame = Image.open("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-FL2VA/first.png")
94
+ last_frame = Image.open("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-FL2VA/last.png")
95
+ prompt = "室内家庭争吵短剧场景,竖屏短剧质感,真实真人表演,中式家庭/小饭馆室内环境,暖色灯光,背景有红色装饰和书法字幅,浅景深,情绪强烈,剪辑节奏紧凑。表演要求:真实短剧表演风格,不要夸张舞台腔。男人的语气是愤怒、委屈、急切的反驳,他说“你到底想干什么?”;中老年女性的语气是尖锐、强势、咄咄���人的质问,她说“你必须赔钱!”。两人之间有强烈对峙感,节奏逐步升级。画面风格:竖屏9:16,手机短剧质感,真人实拍感,浅景深,室内暖光,中近景为主,频繁正反打剪辑,背景保持生活化,不要科幻、不要古装、不要动画感。画面中不要出现任何字幕、文字、平台水印或贴片。 "
96
+ video, audio = pipe(
97
+ prompt=prompt,
98
+ height=832, width=480, num_frames=124, num_inference_steps=50, seed=0,
99
+ keyframes=[first_frame, last_frame], keyframe_indices=[0, -1],
100
+ )
101
+ write_video_audio(
102
+ video=video, audio=audio,
103
+ output_path="fl2va.mp4", fps=24, audio_sample_rate=32000,
104
+ )
105
+ ```
106
+
107
+ ### Ref2VA — 参考图像/视频/音频 → 视频+音频
108
+
109
+ 支持四种参考类型,可在一个列表里组合(`video` 无声,带声视频用 `video_audio`):
110
+
111
+ ```python
112
+ {"type": "image", "image": PIL.Image}
113
+ {"type": "video", "video": list[PIL.Image]} # 无声
114
+ {"type": "audio", "audio": Tensor[C, L], "sample_rate": int}
115
+ {"type": "video_audio", "video": list[PIL.Image], "audio": Tensor[C, L], "sample_rate": int}
116
+ ```
117
+
118
+ ```python
119
+ import torch
120
+ from PIL import Image
121
+ from diffsynth.pipelines.minimax_h3_audio_video import MiniMaxH3Pipeline, ModelConfig
122
+ from diffsynth.utils.data.audio_video import write_video_audio
123
+ from diffsynth.utils.data.audio import read_audio
124
+ from diffsynth.utils.data import VideoData
125
+ from modelscope import dataset_snapshot_download
126
+
127
+ def align_frame_count(frame_count):
128
+ current = max(int(frame_count), 1)
129
+ while current % 17 != 5:
130
+ current += 1
131
+ return current
132
+
133
+ def read_video_with_fps(path, num_out_frames, height, width, fps=24):
134
+ video = VideoData(path, height=height, width=width)
135
+ frames = video.raw_data()
136
+ src_fps = float(video.data.reader.get_meta_data()["fps"])
137
+ out = []
138
+ for k in range(num_out_frames):
139
+ idx = int(round(k * src_fps / fps))
140
+ if idx >= len(frames):
141
+ break
142
+ out.append(frames[idx])
143
+ return out
144
+
145
+ vram_config = {
146
+ "offload_dtype": "disk",
147
+ "offload_device": "disk",
148
+ "onload_dtype": torch.bfloat16,
149
+ "onload_device": "cpu",
150
+ "preparing_dtype": torch.bfloat16,
151
+ "preparing_device": "cuda",
152
+ "computation_dtype": torch.bfloat16,
153
+ "computation_device": "cuda",
154
+ }
155
+ pipe = MiniMaxH3Pipeline.from_pretrained(
156
+ torch_dtype=torch.bfloat16,
157
+ device="cuda",
158
+ model_configs=[
159
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-ref2va-nf4.safetensors", **vram_config),
160
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-text-encoder-nf4.safetensors", **vram_config),
161
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="video_vae_nf4.safetensors", **vram_config),
162
+ ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="audio_vae_nf4.safetensors", **vram_config),
163
+ ],
164
+ processor_config=ModelConfig(model_id="MiniMax/MiniMax-H3", origin_file_pattern="Ref2VA/processor/"),
165
+ vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 5,
166
+ )
167
+
168
+ # Text + Reference Image -> Video + Audio
169
+ dataset_snapshot_download(dataset_id="DiffSynth-Studio/diffsynth_example_dataset", local_dir="data/diffsynth_example_dataset", allow_file_pattern="minimax_h3/MiniMax-H3-Ref2VA/*")
170
+ ref_image = Image.open("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/0.png").convert("RGB")
171
+ prompt = "一个网站页面,网站页面UI设计,网站动效,视频展示了流畅的网页向下滚动效果。一个极具爆发力与动感的产品官网风格产品落地页 UI/UX 演示视频,核心展示主体是该产品图片1。页面采用粗犷有力、倾斜的超大号无衬线字体进行张扬的排版。背景有极具速度感的动态光影、暗色碳纤维或运动透气网眼纹理在交织变换。视频展示了节奏紧凑、充满力量感的网页向下滚动效果,以及鼠标悬停时强烈的视觉放大与颜色反转等 UI 交互动作。"
172
+ video, audio = pipe(
173
+ prompt=prompt,
174
+ height=480, width=832, num_frames=124, num_inference_steps=50, seed=42,
175
+ references=[{"type": "image", "image": Image.open("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/0.png").convert("RGB")}]
176
+ )
177
+ write_video_audio(
178
+ video=video, audio=audio,
179
+ output_path="ti2va.mp4", fps=24, audio_sample_rate=32000,
180
+ )
181
+
182
+ # Text + Reference Audio + Reference Video -> Video + Audio
183
+ ref_video = read_video_with_fps("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/video.mp4", 124, 480, 832)
184
+ ref_audio, sample_rate = read_audio("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/voice.mp3", duration=len(ref_video) / 24, resample=True, resample_rate=pipe.audio_vae.sample_rate)
185
+ prompt = "subject_definitions:\n<Subject 1> is the young man with short wavy blonde hair, wearing a bright pink suit jacket, matching pink trousers, an unbuttoned white shirt, and silver rings, holding a small black lamb in his arms in <Video 1>.\n<Video 1> is the source video for the editing task.\n<Audio 1> is the synchronized audio track of <Video 1>, providing the background music.\n<Audio 2> is the voice timbre reference for <Subject 1>'s voice, containing a spoken male voiceover.\n\nsummary:\n[video editing + audio reference + audio reuse] The target video is an edited version of <Video 1>. <Subject 1>, wearing a bright pink suit and holding a black lamb, stands in a grassy field with other white lambs in the background. The edit animates <Subject 1>'s face to speak the user-provided dialogue. <Audio 1> is partially reused as the continuous background music, while the target references the calm male voice timbre of <Audio 2> for <Subject 1>'s spoken lines.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1]): fully_preserved - the man retains his identity, wavy blonde hair, pink suit, white shirt, accessories, and the black lamb he holds, with his mouth newly animated to speak.\n<Video 1> (source video editing): fully_preserved - the original camera framing, warm golden hour lighting, grassy hill setting, and background white lambs are maintained while the central character is edited.\n<Audio 1>: partially_copy - the atmospheric background music from <Audio 1> is reused in the target video, mixed beneath the newly added spoken dialogue.\n<Audio 2>: reference - the target audio references the male voice timbre from <Audio 2> to generate <Subject 1>'s spoken dialogue.\n\ndetailed_description:\nThe target video is in realistic photographic style.\n[Shot 1] The shot begins from the source <Video 1>, showing <Subject 1>, a young man with short wavy blonde hair, wearing a bright pink suit jacket, matching pink trousers, and a casually unbuttoned white shirt. He stands confidently in a sunlit green pasture, gently holding a small black lamb securely in his arms. The warm, golden hour lighting casts soft shadows across his face and the bright pink fabric of his suit. Behind him, several white lambs stand and graze on the rolling grassy hill against a clear, pale blue sky. The atmospheric background music from <Audio 1> plays continuously throughout the scene. <Subject 1> physically speaks, his mouth movements naturally syncing to the new dialogue, with his voice timbre referencing the calm male delivery from <Audio 2>. Looking thoughtfully forward, <Subject 1> (S1) speaks softly, <d>[English] Follow the wind, live free.</d> As he delivers the line, he subtly shifts his weight, cradling the resting black lamb while the camera slowly pushes in. <Subject 1> (S1) continues his thought, <d>[English] Leave worries behind, enjoy the moment.</d> Exactly as his voice stops, his lips meet in a relaxed, peaceful smile, and his jaw ceases speaking motion. He then turns his gaze slightly away toward the horizon, gently stroking the black lamb's fleece with his fingers as the camera holds on this tranquil, sunlit state through the end of the video.\n\noverall_soundscape:\nThe soundscape consists of the continuous, atmospheric background music from <Audio 1>, overlaid with the clear, calm male dialogue spoken by the main character, referencing the voice timbre of <Audio 2>.\n\nnon_diegetic_music:\nThe atmospheric, sustained background music from <Audio 1> is reused as the continuous score, playing quietly beneath the spoken dialogue."
186
+ video, audio = pipe(
187
+ prompt=prompt,
188
+ height=480, width=832, num_frames=124, num_inference_steps=50, seed=42,
189
+ references=[
190
+ {"type": "video", "video": ref_video},
191
+ {"type": "audio", "audio": ref_audio, "sample_rate": sample_rate},
192
+ ],
193
+ )
194
+ write_video_audio(
195
+ video=video, audio=audio,
196
+ output_path="tav2va.mp4", fps=24, audio_sample_rate=32000,
197
+ )
198
+ ```
199
+
200
+ ## 常用参数
201
+
202
+ - `height` / `width`:分辨率,如 `480x832`(横)或 `832x480`(竖)。
203
+ - `num_frames`:帧数,需满足 `num_frames % 17 == 5`(如 124)。
204
+ - `num_inference_steps`:去噪步数,示例用 50。
205
+ - `keyframes` / `keyframe_indices`:FL2VA 首尾帧控制(`[0, -1]` 表示首帧和尾帧)。
206
+ - `references`:Ref2VA 参考列表,元素为 `{"type": "image|video|audio|video_audio", ...}`。
207
+ - 输出:`write_video_audio(video, audio, output_path, fps=24, audio_sample_rate=32000)`。
208
+
209
+ ## 参考示例脚本
210
+
211
+ 仓库内完整可运行脚本(本 README 的示例即取自这些脚本):
212
+
213
+ - `examples/minimax_h3/model_inference_low_vram/MiniMax-H3-NF4-FL2VA.py`
214
+ - `examples/minimax_h3/model_inference_low_vram/MiniMax-H3-NF4-Ref2VA.py`
215
+ - `examples/minimax_h3/model_inference/MiniMax-H3-NF4-FL2VA.py`(CPU offload)
216
+ - `examples/minimax_h3/model_inference/MiniMax-H3-NF4-Ref2VA.py`(CPU offload)
audio_vae_nf4.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:759662130ba3618b7f196da8f983f857a1f1ec6af8110d657796d9792c0d64e5
3
+ size 284004112
configuration.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"task":"text-to-video-synthesis"}
minimax-h3-fl2va-nf4.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9416f2deb3a6bf1455457bdc373023f266bddcc8646edf7a73394ecd9f13e960
3
+ size 17162138303
minimax-h3-ref2va-nf4.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:765f82272ad5b1e486beef4a5aca719430c6295f3d8d8d07de602a2a740253bb
3
+ size 17162138284
minimax-h3-text-encoder-nf4.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0964f0634856e68522aa64b801afcd31b4e878619eff21a82324e2cd61690f13
3
+ size 15324775807
video_vae_nf4.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6d0cb4ff02ebb74cc6bca40018e6efae5082ccd7eb066fa1263098c6dbf8f6f1
3
+ size 1613201536