MiniMax-H3-NF4
This model is the NF4 quantized version of the video generation model MiniMax-H3. It utilizes the bitsandbytes 4-bit quantization scheme and is designed to be used with DiffSynth-Studio, enabling model inference on devices with limited VRAM and RAM.
Environment Setup
git clone https://github.com/modelscope/DiffSynth-Studio.git
cd DiffSynth-Studio
pip install -e ".[all]"
Inference Code
Enable VRAM Management
Run the following code to perform inference using DiffSynth-Studio. VRAM management will be automatically enabled. The actual VRAM usage depends on the available VRAM on your GPU; a minimum of 8GB VRAM is required to run.
FL2VA (Text-to-Video/Audio):
import torch
from diffsynth.pipelines.minimax_h3_audio_video import MiniMaxH3Pipeline, ModelConfig
from diffsynth.utils.data.audio_video import write_video_audio
from modelscope import dataset_snapshot_download
from PIL import Image
vram_config = {
"offload_dtype": "disk",
"offload_device": "disk",
"onload_dtype": torch.bfloat16,
"onload_device": "cpu",
"preparing_dtype": torch.bfloat16,
"preparing_device": "cuda",
"computation_dtype": torch.bfloat16,
"computation_device": "cuda",
}
pipe = MiniMaxH3Pipeline.from_pretrained(
torch_dtype=torch.bfloat16,
device="cuda",
model_configs=[
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-fl2va-nf4.safetensors", **vram_config),
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-text-encoder-nf4.safetensors", **vram_config),
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="video_vae_nf4.safetensors", **vram_config),
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="audio_vae_nf4.safetensors", **vram_config),
],
processor_config=ModelConfig(model_id="MiniMax/MiniMax-H3", origin_file_pattern="FL2VA/processor/"),
vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 4,
)
prompt = "A girl is very happy, she is speaking in english: “I enjoy working with Diffsynth-Studio, it's a perfect framework.”"
video, audio = pipe(
prompt=prompt,
height=480, width=832, num_frames=124, num_inference_steps=50, seed=0,
)
write_video_audio(
video=video, audio=audio,
output_path="t2va.mp4", fps=24, audio_sample_rate=32000,
)
Ref2VA (Reference-to-Video/Audio):
Expand Code
import torch
from PIL import Image
from diffsynth.pipelines.minimax_h3_audio_video import MiniMaxH3Pipeline, ModelConfig
from diffsynth.utils.data.audio_video import write_video_audio
from diffsynth.utils.data.audio import read_audio
from diffsynth.utils.data import VideoData
from modelscope import dataset_snapshot_download
def align_frame_count(frame_count):
current = max(int(frame_count), 1)
while current % 17 != 5:
current += 1
return current
def read_video_with_fps(path, num_out_frames, height, width, fps=24):
video = VideoData(path, height=height, width=width)
frames = video.raw_data()
src_fps = float(video.data.reader.get_meta_data()["fps"])
out = []
for k in range(num_out_frames):
idx = int(round(k * src_fps / fps))
if idx >= len(frames):
break
out.append(frames[idx])
return out
vram_config = {
"offload_dtype": "disk",
"offload_device": "disk",
"onload_dtype": torch.bfloat16,
"onload_device": "cpu",
"preparing_dtype": torch.bfloat16,
"preparing_device": "cuda",
"computation_dtype": torch.bfloat16,
"computation_device": "cuda",
}
pipe = MiniMaxH3Pipeline.from_pretrained(
torch_dtype=torch.bfloat16,
device="cuda",
model_configs=[
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-ref2va-nf4.safetensors", **vram_config),
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-text-encoder-nf4.safetensors", **vram_config),
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="video_vae_nf4.safetensors", **vram_config),
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="audio_vae_nf4.safetensors", **vram_config),
],
processor_config=ModelConfig(model_id="MiniMax/MiniMax-H3", origin_file_pattern="Ref2VA/processor/"),
vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 5,
)
# Text + Reference Image -> Video + Audio
dataset_snapshot_download(dataset_id="DiffSynth-Studio/diffsynth_example_dataset", local_dir="data/diffsynth_example_dataset", allow_file_pattern="minimax_h3/MiniMax-H3-Ref2VA/*")
ref_image = Image.open("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/0.png").convert("RGB")
prompt = "A website page, website UI design, website animation, video showing smooth webpage scrolling effect. A highly explosive and dynamic product official website style product landing page UI/UX demo video, the core display subject is product image 1. The page uses bold, powerful, tilted oversized sans-serif fonts for flamboyant typography. The background features dynamic light and shadow with extreme speed sense, dark carbon fiber or sports breathable mesh textures interweaving and changing. The video shows a tight-paced, powerful webpage downward scrolling effect, as well as strong visual zoom and color inversion UI interaction actions when hovering the mouse."
video, audio = pipe(
prompt=prompt,
height=480, width=832, num_frames=124, num_inference_steps=50, seed=42,
references=[{"type": "image", "image": Image.open("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/0.png").convert("RGB")}]
)
write_video_audio(
video=video, audio=audio,
output_path="ti2va.mp4", fps=24, audio_sample_rate=32000,
)
# Text + Reference Audio + Reference Video -> Video + Audio
ref_video = read_video_with_fps("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/video.mp4", 124, 480, 832)
ref_audio, sample_rate = read_audio("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/voice.mp3", duration=len(ref_video) / 24, resample=True, resample_rate=pipe.audio_vae.sample_rate)
prompt = "subject_definitions:\n<Subject 1> is the young man with short wavy blonde hair, wearing a bright pink suit jacket, matching pink trousers, an unbuttoned white shirt, and silver rings, holding a small black lamb in his arms in <Video 1>.\n<Video 1> is the source video for the editing task.\n<Audio 1> is the synchronized audio track of <Video 1>, providing the background music.\n<Audio 2> is the voice timbre reference for <Subject 1>'s voice, containing a spoken male voiceover.\n\nsummary:\n[video editing + audio reference + audio reuse] The target video is an edited version of <Video 1>. <Subject 1>, wearing a bright pink suit and holding a black lamb, stands in a grassy field with other white lambs in the background. The edit animates <Subject 1>'s face to speak the user-provided dialogue. <Audio 1> is partially reused as the continuous background music, while the target references the calm male voice timbre of <Audio 2> for <Subject 1>'s spoken lines.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1]): fully_preserved - the man retains his identity, wavy blonde hair, pink suit, white shirt, accessories, and the black lamb he holds, with his mouth newly animated to speak.\n<Video 1> (source video editing): fully_preserved - the original camera framing, warm golden hour lighting, grassy hill setting, and background white lambs are maintained while the central character is edited.\n<Audio 1>: partially_copy - the atmospheric background music from <Audio 1> is reused in the target video, mixed beneath the newly added spoken dialogue.\n<Audio 2>: reference - the target audio references the male voice timbre from <Audio 2> to generate <Subject 1>'s spoken dialogue.\n\ndetailed_description:\nThe target video is in realistic photographic style.\n[Shot 1] The shot begins from the source <Video 1>, showing <Subject 1>, a young man with short wavy blonde hair, wearing a bright pink suit jacket, matching pink trousers, and a casually unbuttoned white shirt. He stands confidently in a sunlit green pasture, gently holding a small black lamb securely in his arms. The warm, golden hour lighting casts soft shadows across his face and the bright pink fabric of his suit. Behind him, several white lambs stand and graze on the rolling grassy hill against a clear, pale blue sky. The atmospheric background music from <Audio 1> plays continuously throughout the scene. <Subject 1> physically speaks, his mouth movements naturally syncing to the new dialogue, with his voice timbre referencing the calm male delivery from <Audio 2>. Looking thoughtfully forward, <Subject 1> (S1) speaks softly, <d>[English] Follow the wind, live free.</d> As he delivers the line, he subtly shifts his weight, cradling the resting black lamb while the camera slowly pushes in. <Subject 1> (S1) continues his thought, <d>[English] Leave worries behind, enjoy the moment.</d> Exactly as his voice stops, his lips meet in a relaxed, peaceful smile, and his jaw ceases speaking motion. He then turns his gaze slightly away toward the horizon, gently stroking the black lamb's fleece with his fingers as the camera holds on this tranquil, sunlit state through the end of the video.\n\noverall_soundscape:\nThe soundscape consists of the continuous, atmospheric background music from <Audio 1>, overlaid with the clear, calm male dialogue spoken by the main character, referencing the voice timbre of <Audio 2>.\n\nnon_diegetic_music:\nThe atmospheric, sustained background music from <Audio 1> is reused as the continuous score, playing quietly beneath the spoken dialogue."
video, audio = pipe(
prompt=prompt,
height=480, width=832, num_frames=124, num_inference_steps=50, seed=42,
references=[
{"type": "video", "video": ref_video},
{"type": "audio", "audio": ref_audio, "sample_rate": sample_rate},
],
)
write_video_audio(
video=video, audio=audio,
output_path="tav2va.mp4", fps=24, audio_sample_rate=32000,
)
Extreme Hardware Optimization
If your computing device has extremely limited performance, we support enabling direct disk-to-VRAM loading. With this configuration, tensors in the model are loaded from disk to VRAM one by one according to the computation order. This allows the model to run with only 8GB of RAM:
vram_config = {
+ "offload_dtype": "disk",
+ "offload_device": "disk",
+ "onload_dtype": "disk",
+ "onload_device": "disk",
+ "preparing_dtype": "disk",
+ "preparing_device": "disk",
+ "computation_dtype": torch.bfloat16,
+ "computation_device": "cuda",
}
pipe = MiniMaxH3Pipeline.from_pretrained(
torch_dtype=torch.bfloat16,
device="cuda",
model_configs=...,
processor_config=...,
+ vram_limit=0,
)
We also support running model inference on Mac M-series chips, although this is not recommended:
vram_config = {
+ "offload_dtype": "disk",
+ "offload_device": "disk",
+ "onload_dtype": "disk",
+ "onload_device": "disk",
+ "preparing_dtype": "disk",
+ "preparing_device": "disk",
+ "computation_dtype": torch.bfloat16,
+ "computation_device": "mps",
}
pipe = MiniMaxH3Pipeline.from_pretrained(
torch_dtype=torch.bfloat16,
+ device="mps",
model_configs=...,
processor_config=...,
+ vram_limit=0,
)
Training Code
This quantized model supports LoRA training. Please follow the steps below to start the training program.
Download the sample dataset:
modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "minimax_h3/MiniMax-H3-FL2VA/*" --local_dir ./data/diffsynth_example_dataset
Training configuration suitable for Data Center GPUs (e.g., Nvidia H20): Run the following script to start the LoRA training program. Requires 48GB VRAM.
accelerate launch examples/minimax_h3/model_training/train.py \
--dataset_base_path data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-FL2VA \
--dataset_metadata_path data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-FL2VA/metadata.csv \
--data_file_keys "video,input_audio" \
--extra_inputs "input_audio" \
--height 480 \
--width 832 \
--num_frames 124 \
--dataset_repeat 100 \
--model_id_with_origin_paths "DiffSynth-Studio/MiniMax-H3-NF4:minimax-h3-text-encoder-nf4.safetensors,DiffSynth-Studio/MiniMax-H3-NF4:minimax-h3-fl2va-nf4.safetensors,DiffSynth-Studio/MiniMax-H3-NF4:video_vae_nf4.safetensors,DiffSynth-Studio/MiniMax-H3-NF4:audio_vae_nf4.safetensors" \
--learning_rate 1e-4 \
--num_epochs 5 \
--remove_prefix_in_ckpt "pipe.dit." \
--output_path "./models/train/MiniMax-H3-T2VA-nf4" \
--lora_base_model "dit" \
--lora_target_modules "qkv_proj,out_proj" \
--lora_rank 32 \
--use_gradient_checkpointing \
--find_unused_parameters
Training configuration suitable for Consumer GPUs (e.g., Nvidia RTX 4090): Run the following scripts to start two-stage split training with gradient checkpointing offload. Requires 24GB VRAM.
accelerate launch examples/minimax_h3/model_training/train.py \
--dataset_base_path data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-FL2VA \
--dataset_metadata_path data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-FL2VA/metadata.csv \
--data_file_keys "video,input_audio" \
--extra_inputs "input_audio" \
--height 480 \
--width 832 \
--num_frames 124 \
--dataset_repeat 1 \
--model_id_with_origin_paths "DiffSynth-Studio/MiniMax-H3-NF4:minimax-h3-text-encoder-nf4.safetensors,DiffSynth-Studio/MiniMax-H3-NF4:video_vae_nf4.safetensors,DiffSynth-Studio/MiniMax-H3-NF4:audio_vae_nf4.safetensors" \
--learning_rate 1e-4 \
--num_epochs 1 \
--remove_prefix_in_ckpt "pipe.dit." \
--output_path "./models/train/MiniMax-H3-T2VA-nf4-split-cache" \
--lora_base_model "dit" \
--lora_target_modules "qkv_proj,out_proj" \
--lora_rank 32 \
--use_gradient_checkpointing \
--use_gradient_checkpointing_offload \
--task "sft:data_process"
accelerate launch examples/minimax_h3/model_training/train.py \
--dataset_base_path "./models/train/MiniMax-H3-T2VA-nf4-split-cache" \
--data_file_keys "video,input_audio" \
--extra_inputs "input_audio" \
--height 480 \
--width 832 \
--num_frames 124 \
--dataset_repeat 100 \
--model_id_with_origin_paths "DiffSynth-Studio/MiniMax-H3-NF4:minimax-h3-fl2va-nf4.safetensors" \
--learning_rate 1e-4 \
--num_epochs 5 \
--remove_prefix_in_ckpt "pipe.dit." \
--output_path "./models/train/MiniMax-H3-T2VA-nf4" \
--lora_base_model "dit" \
--lora_target_modules "qkv_proj,out_proj" \
--lora_rank 32 \
--use_gradient_checkpointing \
--use_gradient_checkpointing_offload \
--find_unused_parameters \
--task "sft:train"
References
- DiffSynth-Studio Documentation: Minimax-H3
- DiffSynth-Studio Documentation: VRAM Management
- DiffSynth-Studio Documentation: Two-Stage Split Training
- DiffSynth-Studio Documentation: Low VRAM Training
Model tree for DiffSynth-Studio/MiniMax-H3-NF4
Base model
MiniMaxAI/MiniMax-H3