diff --git a/.gitignore b/.gitignore index 13c5fbc871..88035e8c2b 100644 --- a/.gitignore +++ b/.gitignore @@ -142,3 +142,4 @@ fastvideo/tests/ssim/reference_videos/** *.nvimlog .nvimlog .python-version +scripts/benchmarks/minimax_h3_pro6000/headline_results/ diff --git a/docs/assets/cookbook-recipes.json b/docs/assets/cookbook-recipes.json index e89e58d121..5f26fe1bb0 100644 --- a/docs/assets/cookbook-recipes.json +++ b/docs/assets/cookbook-recipes.json @@ -1,5 +1,5 @@ { - "version": 12, + "version": 14, "recipes": [ { "id": "fastwan21-t2v", @@ -601,6 +601,77 @@ "Height, width, frames, and steps in the YAML are examples. Edit them or pass CLI flags. See docs/getting_started/installation/spark_pair.md." ] }, + { + "id": "compacth3-rtx5090", + "group": "compacth3-rtx5090", + "group_label": "CompactH3 on RTX 5090", + "group_task": "4-step 42-block text to video + audio", + "family": "minimax_h3", + "stage": "inference", + "task": "Few-step text to video (with audio)", + "label": "CompactH3 NVFP4 on RTX 5090", + "summary": "Run the 42-block CompactH3 NVFP4 DiT with the NVFP4 Qwen3-VL encoder, Comfy int8-convrot VAE, and SageAttention3 FP4 on one 32 GB RTX 5090. Sequential load parks the encoder in pinned host RAM.", + "model": "./CompactH3", + "source": "examples/inference/basic/basic_compacth3_rtx5090.yaml", + "serving": { + "source": "examples/serving/openai_compacth3_rtx5090.yaml", + "install": "UV_TORCH_BACKEND=cu128 uv pip install -e \".[fasth3]\"", + "prepare": "hf download FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree --local-dir ./CompactH3 --include model_index.json --include \"tokenizer/**\" --include \"processor/**\" --include \"scheduler/**\" --include \"audio_scheduler/**\" --include \"audio_vae/**\" --include \"vae/**\"\nhf download aryan5v/FastH3-20B-42block-dmd2-ckpt1400-bf16 --local-dir ./CompactH3/transformer\nhf download aryan5v/FastH3-20B-42block-dmd2-ckpt1400-nvfp4 --local-dir ./CompactH3/transformer --include nvfp4_weights.safetensors\nhf download KyleNeverGivesUp/FastH3-text-encoder-nvfp4 --local-dir ./CompactH3/text_encoder\nhf download Comfy-Org/MiniMax-H3 --local-dir ./Comfy-MiniMax-H3 --include vae/minimax_h3_video_vae_int8_convrot.safetensors\ncp ./Comfy-MiniMax-H3/vae/minimax_h3_video_vae_int8_convrot.safetensors ./CompactH3/vae/", + "env": "FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FLASHINFER_CUDA_ARCH_LIST=12.0a" + }, + "command": "FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FLASHINFER_CUDA_ARCH_LIST=12.0a FASTVIDEO_STAGE_LOGGING=1 fastvideo generate --config examples/inference/basic/basic_compacth3_rtx5090.yaml", + "gpu_types": ["NVIDIA"], + "hardware": { + "platform": "cuda", + "gpu_count": 1, + "evidence": "source-configured" + }, + "evidence": "Source-backed", + "expected_artifact": "MP4 under outputs/compacth3_rtx5090/", + "modes": ["T2VA", "CompactH3 NVFP4", "RTX 5090"], + "limitations": [ + "Assemble ./CompactH3 before running. The DiT NVFP4 export and encoder snapshot are gated; run huggingface-cli login and accept each repo license.", + "32 GB cannot keep the NVFP4 encoder and DiT on the GPU together. Keep h3_sequential_load on and lazy_module_load off.", + "Blackwell sm_120 needs ATTN_QAT_INFER, FLASHINFER_CUDA_ARCH_LIST=12.0a, and a CUDA 12.8 PyTorch wheel.", + "Legal num_frames values are 17n+5, capped at 362 (15.08 s). Native 16:9 sizes include 832x480 and 1344x768; 1344x768 on 32 GB is unmeasured." + ] + }, + { + "id": "compacth3-rtx-pro6000", + "group": "compacth3-rtx-pro6000", + "group_label": "CompactH3 on RTX PRO 6000", + "group_task": "4-step 42-block text to video + audio", + "family": "minimax_h3", + "stage": "inference", + "task": "Few-step text to video (with audio)", + "label": "CompactH3 NVFP4 on RTX PRO 6000 Blackwell", + "summary": "Run CompactH3 NVFP4 with the encoder, DiT, and int8-convrot VAE resident on one 96 GB RTX PRO 6000 Blackwell. The checked-in example is 1344x768 and 124 frames (5.17 s) with VAE torch.compile. Use 832x480 for clip-queue playground traffic.", + "model": "./CompactH3", + "source": "examples/inference/basic/basic_compacth3_rtx_pro6000.yaml", + "serving": { + "source": "examples/serving/openai_compacth3_rtx_pro6000.yaml", + "install": "UV_TORCH_BACKEND=cu128 uv pip install -e \".[fasth3]\"", + "prepare": "hf download FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree --local-dir ./CompactH3 --include model_index.json --include \"tokenizer/**\" --include \"processor/**\" --include \"scheduler/**\" --include \"audio_scheduler/**\" --include \"audio_vae/**\" --include \"vae/**\"\nhf download aryan5v/FastH3-20B-42block-dmd2-ckpt1400-bf16 --local-dir ./CompactH3/transformer\nhf download aryan5v/FastH3-20B-42block-dmd2-ckpt1400-nvfp4 --local-dir ./CompactH3/transformer --include nvfp4_weights.safetensors\nhf download KyleNeverGivesUp/FastH3-text-encoder-nvfp4 --local-dir ./CompactH3/text_encoder\nhf download Comfy-Org/MiniMax-H3 --local-dir ./Comfy-MiniMax-H3 --include vae/minimax_h3_video_vae_int8_convrot.safetensors\ncp ./Comfy-MiniMax-H3/vae/minimax_h3_video_vae_int8_convrot.safetensors ./CompactH3/vae/", + "env": "FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FLASHINFER_CUDA_ARCH_LIST=12.0a" + }, + "command": "FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FLASHINFER_CUDA_ARCH_LIST=12.0a FASTVIDEO_STAGE_LOGGING=1 fastvideo generate --config examples/inference/basic/basic_compacth3_rtx_pro6000.yaml", + "gpu_types": ["NVIDIA"], + "hardware": { + "platform": "cuda", + "gpu_count": 1, + "evidence": "source-configured" + }, + "evidence": "Source-backed", + "expected_artifact": "MP4 under outputs/compacth3_rtx_pro6000/", + "modes": ["T2VA", "CompactH3 NVFP4", "RTX PRO 6000"], + "limitations": [ + "Assemble ./CompactH3 before running. The DiT NVFP4 export and encoder snapshot are gated; run huggingface-cli login and accept each repo license.", + "96 GB keeps the encoder, DiT, and VAE on GPU. Do not enable h3_sequential_load or lazy_module_load on this box.", + "Enable compile.vae_enabled. Leave inference_torch_compile off: FlashInfer and Sage3 custom ops cannot be compiled.", + "Blackwell sm_120 needs ATTN_QAT_INFER, FLASHINFER_CUDA_ARCH_LIST=12.0a, and a CUDA 12.8 PyTorch wheel.", + "Legal num_frames values are 17n+5, capped at 362 (15.08 s). Native 16:9 sizes include 832x480 and 1344x768. Dense CompactH3 has zero VSA gates; the pipeline raises if VIDEO_SPARSE_ATTN_H3 is loaded with all-zero to_gate_compress weights." + ] + }, { "id": "fasth3-8step-v2-cuda", "group": "fasth3-8step-v2", diff --git a/docs/contributing/env_vars.md b/docs/contributing/env_vars.md index 4f47ddcb1f..883e2d8ac8 100644 --- a/docs/contributing/env_vars.md +++ b/docs/contributing/env_vars.md @@ -240,6 +240,35 @@ longer exists also fails the test, so the fixing pull request deletes its entry. | `FASTVIDEO_VBENCH_FULL_INFO_JSON` | str | unset | eval | Path to VBench_full_info.json, used instead of the vendored copy. Deprecated names: `VBENCH_FULL_INFO_JSON`. | | `FASTVIDEO_FVD_REF_FEATURES` | str | unset | eval | Cached reference-feature file for the FVD metric. | | `FASTVIDEO_FAD_REF_FEATURES` | str | unset | eval | Cached reference-feature file for the audio Frechet distance metric. | +| `FASTVIDEO_H3_VSA_FP4` | bool | `0` | attention | Run MiniMax-H3 VSA attention on the block-sparse SageAttention3 FP4 kernel (sm_120, no-grad, single sequence-parallel rank). | +| `FASTVIDEO_H3_VSA_TILE_FIRST` | bool | `0` | attention | Single-rank MiniMax-H3 VSA with one tile gather of the block input instead of separate Q/K/V/gate scatters. | +| `FASTVIDEO_H3_VSA_SM89_KERNEL` | one of original, bf16, int8 | `original` | attention | Fine-attention kernel for MiniMax-H3 VSA on sm_89: original, bf16, or int8 (INT8 QK, BF16 PV). | +| `FASTVIDEO_H3_SIM_SP_FP8` | bool | `0` | debug | Simulate the FP8 sequence-parallel exchange of the MiniMax-H3 FP4 VSA path on one rank. | +| `FASTVIDEO_H3_FFN_CHUNK_TOKENS` | int | `0` | performance | Inference-only MiniMax-H3 FFN token chunk size; 0 runs the FFN unchunked. | +| `FASTVIDEO_H3_FP8_ATTENTION` | bool | `0` | performance | With NVFP4 layer_profile h3_dit_ffn, run MiniMax-H3 attention projections in FP8. | +| `FASTVIDEO_H3_FP8_GRANULARITY` | one of tensor, channel | `tensor` | performance | FP8 scaling granularity for FASTVIDEO_H3_FP8_ATTENTION. | +| `FASTVIDEO_NVFP4_MM_BACKEND` | str | `auto` | performance | FlashInfer mm_fp4 backend for NVFP4 linears, e.g. auto or cutlass. | +| `FASTVIDEO_NVFP4_ACT_AMAX` | path | unset | performance | JSON of calibrated NVFP4 input amax per linear, keyed b<block>.<sub> or full prefix; sets a static activation scale. | +| `FASTVIDEO_NVFP4_DYNAMIC_ACT` | str | `""` | performance | NVFP4 linears that derive the activation scale per call: all, or comma-separated layer-name suffixes such as ff.fc_out. | +| `FASTVIDEO_H3_ADALN_CACHE` | bool | `0` | performance | Cache MiniMax-H3 AdaLN modulation per timestep instead of keeping the projection weights resident. | +| `FASTVIDEO_H3_ADALN_TABLE` | path | unset | performance | Precomputed MiniMax-H3 AdaLN modulation table; enables the cache and skips loading the AdaLN projection weights. | +| `FASTVIDEO_H3_ADALN_DUMP` | path | unset | debug | Write the MiniMax-H3 AdaLN modulation table to this path while sampling. | +| `FASTVIDEO_H3_SPLICE_TRANSFORMER` | path | unset | eval | Second MiniMax-H3 transformer that runs the late DMD steps (checkpoint step-splice evaluation). | +| `FASTVIDEO_H3_SPLICE_FROM_STEP` | int | `4` | eval | First denoising step run by FASTVIDEO_H3_SPLICE_TRANSFORMER. | +| `FASTVIDEO_H3_ENCODER_LAYERWISE` | bool | `0` | performance | Stream MiniMax-H3 text-encoder language layers through exact-size pinned host memory (text-only prompts). | +| `FASTVIDEO_H3_ENCODER_FUSED_DEQUANT` | bool | `0` | performance | Expand the serialized NVFP4 MiniMax-H3 text encoder with one fused Triton pass on GPUs without FP4 GEMM. | +| `FASTVIDEO_H3_VAE_TILE_BATCH` | int | `1` | performance | Spatial tiles per MiniMax-H3 video VAE decoder call; 1 decodes per tile. | +| `FASTVIDEO_H3_VAE_INT8_SHARED_QKV` | bool | `0` | performance | Share the INT8 activation rotation and quantization across the MiniMax-H3 VAE Q/K/V projections. | +| `FASTVIDEO_H3_VAE_INT8_TRANSPOSE_VIEW` | bool | `0` | performance | Use transposed weight views in the MiniMax-H3 VAE INT8 projections. | +| `FASTVIDEO_H3_VAE_INT8_FUSED_DEQUANT` | bool | `0` | performance | Fused dequantization epilogue for the MiniMax-H3 VAE INT8 projections. | +| `FASTVIDEO_H3_PINNED_SWAP` | bool | `1` | performance | Swap offloaded MiniMax-H3 modules through exact-size pinned host arenas. | +| `FASTVIDEO_H3_PARK_MODULES` | str | unset | performance | Comma-separated MiniMax-H3 denoise modules parked on the host while the text encoder runs, e.g. vae,audio_vae. | +| `FASTVIDEO_LAYERWISE_OFFLOAD_BUFFERS` | bool | `0` | performance | Layerwise offload also streams large buffers such as packed FP4/FP8 weights. | +| `FASTVIDEO_LAYERWISE_RESIDENT_BLOCKS` | int | `0` | performance | Keep the first N layerwise-offloaded blocks resident on the GPU. | +| `FASTVIDEO_H3_SP_PROFILE` | bool | `0` | profiling | CUDA-event spans per stage over one MiniMax-H3 FP4 VSA DiT forward. | +| `FASTVIDEO_H3_CAPTURE_QKV` | path | unset | debug | Directory for captured real MiniMax-H3 Q/K/V attention inputs. | +| `FASTVIDEO_CUDA_MEMORY_CAP_GIB` | float | `0.0` | debug | Cap this process's CUDA allocator at this many GiB to emulate a smaller GPU; 0 leaves it uncapped. | +| `FASTVIDEO_MEMORY_REPORT` | bool | `0` | debug | Log bytes held per pipeline component by device and dtype after loading. | | `FASTVIDEO_TEST_LTX2_OVERFIT_DATA_DIR` | str | `data/cats` | test | Raw data directory for preprocess_ltx2_overfit.py. Deprecated names: `LTX2_OVERFIT_DATA_DIR`. | | `FASTVIDEO_TEST_LTX2_OVERFIT_CAPTION_JSON` | str | `videos2caption_1_sample.json` | test | Caption file, relative to the raw data directory. Deprecated names: `LTX2_OVERFIT_CAPTION_JSON`. | | `FASTVIDEO_TEST_LTX2_OVERFIT_VIDEO_SUBDIR` | str | `video` | test | Video subdirectory, relative to the raw data directory. Deprecated names: `LTX2_OVERFIT_VIDEO_SUBDIR`. | diff --git a/docs/cookbook/cosmos.md b/docs/cookbook/cosmos.md index fbb813fae6..4d9c8c0fbc 100644 --- a/docs/cookbook/cosmos.md +++ b/docs/cookbook/cosmos.md @@ -5,7 +5,7 @@ hide: # Cosmos recipes -
Primary focus · Inference
Generate video and audio with H3. Run a server on CUDA, one DGX Spark, or Apple Silicon MLX to iterate on prompts, or call the pipeline directly from Python.
- 9 maintained recipes +Generate video and audio with H3. Run a server on CUDA, one Blackwell GPU, one DGX Spark, or Apple Silicon MLX to iterate on prompts, or call the pipeline directly from Python.
+ 11 maintained recipes
CUDA covers T2VA, FL2VA, and Ref2VA on the full checkpoint, plus FastH3
- V1 and FastH3 V2. FastH3 V1 also has a DGX Spark runtime with
+ V1 and FastH3 V2, plus CompactH3 NVFP4 on one Blackwell GPU. FastH3 V1 also has a DGX Spark runtime with
a 1-Spark or 2-Spark device row. MLX is T2VA only: V1 and V2.
Temporal --fast, spatial --fast-spatial, and opt-in VSA are flags on the same
MLX script, not extra recipes.
@@ -64,7 +66,7 @@ is why V2 is the higher-quality FastH3. The V2 schedule contract is in
sp_size=2) over QSFP RoCE. Select NVIDIA DGX Spark, then 1 Spark or 2 Sparks.Choose the result you want, then use a maintained CUDA, DGX Spark, or MLX path. +
Choose the result you want, then use a maintained CUDA, Blackwell, DGX Spark, or MLX path. Device claims stay tied to checked-in sources and recorded runs.
--fast, optional spatial --fast-spatial, and opt-in VSA on --include-vsa checkpoints. FastH3 V2 MLX converts with --include-vsa and runs eight forwards. FL2VA, Ref2VA, and two-pass refinement are not wired.FASTVIDEO_FA4=0 and FASTVIDEO_VSA_SM100A=0. Legal num_frames values are 17n+5, capped at 362 (15.08 s). A 345-frame request on one Spark can OOM.FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER, FASTVIDEO_FA4=0, FASTVIDEO_VSA_SM100A=0, and FLASHINFER_CUDA_ARCH_LIST=12.0a. On PRO 6000 enable VAE compile and leave DiT inference_torch_compile off. CompactH3 is a dense prune; do not enable VIDEO_SPARSE_ATTN_H3 until a VSA-trained student exists.FASTVIDEO_FA4=0 and FASTVIDEO_VSA_SM100A=0. Legal num_frames values are 17n+5, capped at 362 (15.08 s). A 345-frame request on one Spark can OOM. Native 16:9 sizes include 832×480 and 1344×768.huggingface-cli login and confirm you accepted the model's license on Hugging Face.