Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
30 changes: 30 additions & 0 deletions configs/minimax_h3/fp8/minimax_h3_t2av.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,30 @@
{
"infer_steps": 30,
"target_video_length": 124,
"target_height": 544,
"target_width": 960,
"fps": 24,
"target_fps": 24,
"enable_cfg": false,
"cpu_offload": true,
"offload_granularity": "model",
"text_encoder_cpu_offload": true,
"vae_cpu_offload": true,
"lazy_load": false,
"unload_modules": false,
"attn_type": "sage_attn2",
"rms_type": "sgl-kernel",
"rope_type": "minimax_h3_triton_rope",
"feature_caching": "NoCaching",
"use_compile": false,
"video_flow_shift": 12.0,
"audio_flow_shift": 3.0,
"vae_spatial_scale_factor": 16,
"audio_sampling_rate": 32000,
"audio_latents_per_second": 40,
"audio_channels": 2,
"keep_latents_dtype_in_scheduler": true,
"dit_quantized": true,
"dit_quant_scheme": "fp8-sgl",
"dit_quantized_ckpt": "/data/nvme6/gushiqiao/models/MiniMax-H3/quantized/fp8/minimax_h3_fp8.safetensors"
}
27 changes: 27 additions & 0 deletions configs/minimax_h3/minimax_h3_fl2av.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
{
"infer_steps": 30,
"target_video_length": 124,
"target_height": 544,
"target_width": 960,
"fps": 24,
"target_fps": 24,
"enable_cfg": false,
"cpu_offload": true,
"offload_granularity": "model",
"text_encoder_cpu_offload": true,
"vae_cpu_offload": true,
"lazy_load": false,
"unload_modules": false,
"attn_type": "sage_attn2",
"rms_type": "sgl-kernel",
"rope_type": "minimax_h3_triton_rope",
"feature_caching": "NoCaching",
"use_compile": false,
"video_flow_shift": 12.0,
"audio_flow_shift": 3.0,
"vae_spatial_scale_factor": 16,
"audio_sampling_rate": 32000,
"audio_latents_per_second": 40,
"audio_channels": 2,
"keep_latents_dtype_in_scheduler": true
}
27 changes: 27 additions & 0 deletions configs/minimax_h3/minimax_h3_i2av.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
{
"infer_steps": 30,
"target_video_length": 124,
"target_height": 544,
"target_width": 960,
"fps": 24,
"target_fps": 24,
"enable_cfg": false,
"cpu_offload": true,
"offload_granularity": "model",
"text_encoder_cpu_offload": true,
"vae_cpu_offload": true,
"lazy_load": false,
"unload_modules": false,
"attn_type": "sage_attn2",
"rms_type": "sgl-kernel",
"rope_type": "minimax_h3_triton_rope",
"feature_caching": "NoCaching",
"use_compile": false,
"video_flow_shift": 12.0,
"audio_flow_shift": 3.0,
"vae_spatial_scale_factor": 16,
"audio_sampling_rate": 32000,
"audio_latents_per_second": 40,
"audio_channels": 2,
"keep_latents_dtype_in_scheduler": true
}
27 changes: 27 additions & 0 deletions configs/minimax_h3/minimax_h3_l2av.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
{
"infer_steps": 30,
"target_video_length": 124,
"target_height": 544,
"target_width": 960,
"fps": 24,
"target_fps": 24,
"enable_cfg": false,
"cpu_offload": true,
"offload_granularity": "model",
"text_encoder_cpu_offload": true,
"vae_cpu_offload": true,
"lazy_load": false,
"unload_modules": false,
"attn_type": "sage_attn2",
"rms_type": "sgl-kernel",
"rope_type": "minimax_h3_triton_rope",
"feature_caching": "NoCaching",
"use_compile": false,
"video_flow_shift": 12.0,
"audio_flow_shift": 3.0,
"vae_spatial_scale_factor": 16,
"audio_sampling_rate": 32000,
"audio_latents_per_second": 40,
"audio_channels": 2,
"keep_latents_dtype_in_scheduler": true
}
27 changes: 27 additions & 0 deletions configs/minimax_h3/minimax_h3_ref2av.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
{
"infer_steps": 30,
"target_video_length": 124,
"target_height": 544,
"target_width": 960,
"fps": 24,
"target_fps": 24,
"enable_cfg": false,
"cpu_offload": true,
"offload_granularity": "model",
"text_encoder_cpu_offload": true,
"vae_cpu_offload": true,
"lazy_load": false,
"unload_modules": false,
"attn_type": "sage_attn2",
"rms_type": "sgl-kernel",
"rope_type": "minimax_h3_triton_rope",
"feature_caching": "NoCaching",
"use_compile": false,
"video_flow_shift": 12.0,
"audio_flow_shift": 3.0,
"vae_spatial_scale_factor": 16,
"audio_sampling_rate": 32000,
"audio_latents_per_second": 40,
"audio_channels": 2,
"keep_latents_dtype_in_scheduler": true
}
27 changes: 27 additions & 0 deletions configs/minimax_h3/minimax_h3_t2av.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
{
"infer_steps": 30,
"target_video_length": 362,
"target_height": 544,
"target_width": 960,
"fps": 24,
"target_fps": 24,
"enable_cfg": false,
"cpu_offload": true,
"offload_granularity": "model",
"text_encoder_cpu_offload": true,
"vae_cpu_offload": true,
"lazy_load": false,
"unload_modules": false,
"attn_type": "sage_attn2",
"rms_type": "sgl-kernel",
"rope_type": "minimax_h3_triton_rope",
"feature_caching": "NoCaching",
"use_compile": false,
"video_flow_shift": 12.0,
"audio_flow_shift": 3.0,
"vae_spatial_scale_factor": 16,
"audio_sampling_rate": 32000,
"audio_latents_per_second": 40,
"audio_channels": 2,
"keep_latents_dtype_in_scheduler": true
}
28 changes: 28 additions & 0 deletions configs/minimax_h3/minimax_h3_t2av_block_offload.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,28 @@
{
"infer_steps": 30,
"target_video_length": 124,
"target_height": 768,
"target_width": 1344,
"fps": 24,
"target_fps": 24,
"enable_cfg": false,
"cpu_offload": true,
"offload_granularity": "block",
"text_encoder_cpu_offload": true,
"text_encoder_offload_granularity": "block",
"vae_cpu_offload": true,
"lazy_load": false,
"unload_modules": false,
"attn_type": "sage_attn2",
"rms_type": "sgl-kernel",
"rope_type": "minimax_h3_triton_rope",
"feature_caching": "NoCaching",
"use_compile": false,
"video_flow_shift": 12.0,
"audio_flow_shift": 3.0,
"vae_spatial_scale_factor": 16,
"audio_sampling_rate": 32000,
"audio_latents_per_second": 40,
"audio_channels": 2,
"keep_latents_dtype_in_scheduler": true
}
31 changes: 31 additions & 0 deletions configs/minimax_h3/minimax_h3_t2av_sp.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,31 @@
{
"infer_steps": 30,
"target_video_length": 124,
"target_height": 544,
"target_width": 960,
"fps": 24,
"target_fps": 24,
"enable_cfg": false,
"cpu_offload": true,
"offload_granularity": "model",
"text_encoder_cpu_offload": true,
"vae_cpu_offload": true,
"lazy_load": false,
"unload_modules": false,
"attn_type": "sage_attn2",
"rms_type": "sgl-kernel",
"rope_type": "minimax_h3_triton_rope",
"feature_caching": "NoCaching",
"use_compile": false,
"video_flow_shift": 12.0,
"audio_flow_shift": 3.0,
"vae_spatial_scale_factor": 16,
"audio_sampling_rate": 32000,
"audio_latents_per_second": 40,
"audio_channels": 2,
"keep_latents_dtype_in_scheduler": true,
"parallel": {
"seq_p_size": 4,
"seq_p_attn_type": "ulysses"
}
}
31 changes: 31 additions & 0 deletions configs/minimax_h3/minimax_h3_t2av_tp.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,31 @@
{
"infer_steps": 30,
"target_video_length": 124,
"target_height": 544,
"target_width": 960,
"fps": 24,
"target_fps": 24,
"enable_cfg": false,
"cpu_offload": false,
"offload_granularity": "model",
"text_encoder_cpu_offload": false,
"text_encoder_tensor_parallel": true,
"vae_cpu_offload": true,
"lazy_load": false,
"unload_modules": false,
"attn_type": "sage_attn2",
"rms_type": "sgl-kernel",
"rope_type": "minimax_h3_triton_rope",
"feature_caching": "NoCaching",
"use_compile": false,
"video_flow_shift": 12.0,
"audio_flow_shift": 3.0,
"vae_spatial_scale_factor": 16,
"audio_sampling_rate": 32000,
"audio_latents_per_second": 40,
"audio_channels": 2,
"keep_latents_dtype_in_scheduler": true,
"parallel": {
"tensor_p_size": 2
}
}
33 changes: 33 additions & 0 deletions configs/minimax_h3/minimax_h3_t2av_tp_sp.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,33 @@
{
"infer_steps": 30,
"target_video_length": 124,
"target_height": 544,
"target_width": 960,
"fps": 24,
"target_fps": 24,
"enable_cfg": false,
"cpu_offload": false,
"offload_granularity": "model",
"text_encoder_cpu_offload": false,
"text_encoder_tensor_parallel": true,
"vae_cpu_offload": true,
"lazy_load": false,
"unload_modules": false,
"attn_type": "sage_attn2",
"rms_type": "sgl-kernel",
"rope_type": "minimax_h3_triton_rope",
"feature_caching": "NoCaching",
"use_compile": false,
"video_flow_shift": 12.0,
"audio_flow_shift": 3.0,
"vae_spatial_scale_factor": 16,
"audio_sampling_rate": 32000,
"audio_latents_per_second": 40,
"audio_channels": 2,
"keep_latents_dtype_in_scheduler": true,
"parallel": {
"tensor_p_size": 2,
"seq_p_size": 2,
"seq_p_attn_type": "ulysses"
}
}
11 changes: 8 additions & 3 deletions lightx2v/infer.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@
from lightx2v.models.runners.lingbot_video.lingbot_video_runner import LingBotVideoRunner # noqa: F401
from lightx2v.models.runners.longcat_image.longcat_image_runner import LongCatImageRunner # noqa: F401
from lightx2v.models.runners.ltx2.ltx2_runner import LTX2ARRunner, LTX2Runner # noqa: F401
from lightx2v.models.runners.minimax_h3.minimax_h3_runner import MiniMaxH3Runner # noqa: F401
from lightx2v.models.runners.motus.motus_runner import MotusRunner # noqa: F401
from lightx2v.models.runners.neopp.neopp_runner import NeoppRunner # noqa: F401
from lightx2v.models.runners.qwen_image.qwen_image_runner import QwenImageRunner # noqa: F401
Expand Down Expand Up @@ -124,6 +125,7 @@ def main():
"flux2_dev",
"ltx2",
"ltx2_ar",
"minimax_h3",
"bagel",
"sensenova_vision",
"seedvr2",
Expand Down Expand Up @@ -158,6 +160,9 @@ def main():
"rs2v",
"t2av",
"i2av",
"l2av",
"fl2av",
"ref2av",
"i2va",
"v2av",
"ltx2_s2v",
Expand Down Expand Up @@ -186,15 +191,15 @@ def main():
"--image_path",
type=str,
default="",
help="The path to input image file(s), including HunyuanImage3 ti2t/ti2i. Multiple paths should be comma-separated. Example: 'path1.jpg,path2.jpg'",
help="The path to input image file(s), including HunyuanImage3 ti2t/ti2i and MiniMax-H3 ref2av reference images. Multiple paths should be comma-separated. Example: 'path1.jpg,path2.jpg'",
)
parser.add_argument("--state_path", type=str, default="", help="The path to input robot state file for robot i2v/i2va inference.")
parser.add_argument("--last_frame_path", type=str, default="", help="The path to last frame file for first-last-frame-to-video (flf2v) task")
parser.add_argument(
"--audio_path",
type=str,
default="",
help="Input audio path: Wan s2v / rs2v, or required for LTX-2 task ltx2_s2v.",
help="Input audio path: Wan s2v / rs2v, LTX-2 ltx2_s2v, or MiniMax-H3 ref2av reference audio. H3 accepts comma-separated paths.",
)
parser.add_argument("--image_strength", type=str, default="1.0", help="i2av: single float, or comma-separated floats (one per image, or one value broadcast). Example: 1.0 or 1.0,0.85,0.9")
parser.add_argument(
Expand Down Expand Up @@ -306,7 +311,7 @@ def main():
"--video_path",
type=str,
default=None,
help="input video path (for sr / v2v / v2av task). For v2av this is the pre-processed control/reference video (pose / canny / depth / motion-track for motion-transfer, or the degraded source video for ICEdit).",
help="Input video path for sr/v2v/v2av, or MiniMax-H3 ref2av reference video. H3 accepts comma-separated paths. For v2av this is the pre-processed control/reference video (pose / canny / depth / motion-track for motion-transfer, or the degraded source video for ICEdit).",
)
parser.add_argument("--sr_ratio", type=float, default=2.0, help="super resolution ratio for sr task")
parser.add_argument(
Expand Down
1 change: 1 addition & 0 deletions lightx2v/models/audio_encoders/hf/minimax_h3/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
from .audio_vae import AutoencoderKLMiniMaxH3AudioNative, MiniMaxH3AudioVAE
Loading
Loading