Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ name: "minimax-m3-vllm-disagg-b300-1p1d-dep2-tp4-fp4-8k1k-eagle3"

model:
path: "nvidia/MiniMax-M3-NVFP4"
container: "vllm/vllm-openai:nightly-4080263bb2c5d10deac17aaeb88e0823bc35bca9"
container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7"
precision: fp4

resources:
Expand Down Expand Up @@ -50,6 +50,7 @@ backend:
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8"}'
block-size: 128
kv-cache-dtype: fp8
gpu-memory-utilization: 0.95
max-model-len: 9472
language-model-only: true
Expand All @@ -66,8 +67,9 @@ backend:
trust-remote-code: true
no-enable-prefix-caching: true
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8", "minimax_m3_msa_decode_backend": "cutlass"}'
block-size: 128
kv-cache-dtype: fp8
gpu-memory-utilization: 0.95
max-model-len: 9472
language-model-only: true
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ name: "minimax-m3-vllm-disagg-b300-1p2d-dep2-tp4-fp4-8k1k-eagle3"

model:
path: "nvidia/MiniMax-M3-NVFP4"
container: "vllm/vllm-openai:nightly-4080263bb2c5d10deac17aaeb88e0823bc35bca9"
container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7"
precision: fp4

resources:
Expand Down Expand Up @@ -50,6 +50,7 @@ backend:
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8"}'
block-size: 128
kv-cache-dtype: fp8
gpu-memory-utilization: 0.95
max-model-len: 9472
language-model-only: true
Expand All @@ -66,8 +67,9 @@ backend:
trust-remote-code: true
no-enable-prefix-caching: true
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8", "minimax_m3_msa_decode_backend": "cutlass"}'
block-size: 128
kv-cache-dtype: fp8
gpu-memory-utilization: 0.95
max-model-len: 9472
language-model-only: true
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ name: "minimax-m3-vllm-disagg-b300-1p4d-dep2-tp4-fp4-8k1k-eagle3"

model:
path: "nvidia/MiniMax-M3-NVFP4"
container: "vllm/vllm-openai:nightly-4080263bb2c5d10deac17aaeb88e0823bc35bca9"
container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7"
precision: fp4

resources:
Expand Down Expand Up @@ -50,6 +50,7 @@ backend:
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8"}'
block-size: 128
kv-cache-dtype: fp8
gpu-memory-utilization: 0.95
max-model-len: 9472
language-model-only: true
Expand All @@ -66,8 +67,9 @@ backend:
trust-remote-code: true
no-enable-prefix-caching: true
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8", "minimax_m3_msa_decode_backend": "cutlass"}'
block-size: 128
kv-cache-dtype: fp8
gpu-memory-utilization: 0.95
max-model-len: 9472
language-model-only: true
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ name: "minimax-m3-vllm-disagg-b300-1p6d-dep2-tp4-fp4-8k1k-eagle3"

model:
path: "nvidia/MiniMax-M3-NVFP4"
container: "vllm/vllm-openai:nightly-4080263bb2c5d10deac17aaeb88e0823bc35bca9"
container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7"
precision: fp4

resources:
Expand Down Expand Up @@ -50,6 +50,7 @@ backend:
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8"}'
block-size: 128
kv-cache-dtype: fp8
gpu-memory-utilization: 0.95
max-model-len: 9472
language-model-only: true
Expand All @@ -66,8 +67,9 @@ backend:
trust-remote-code: true
no-enable-prefix-caching: true
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8", "minimax_m3_msa_decode_backend": "cutlass"}'
block-size: 128
kv-cache-dtype: fp8
gpu-memory-utilization: 0.95
max-model-len: 9472
language-model-only: true
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ name: "minimax-m3-vllm-disagg-b300-2p3d-dep2-tp4-fp4-8k1k-eagle3"

model:
path: "nvidia/MiniMax-M3-NVFP4"
container: "vllm/vllm-openai:nightly-4080263bb2c5d10deac17aaeb88e0823bc35bca9"
container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7"
precision: fp4

resources:
Expand Down Expand Up @@ -50,6 +50,7 @@ backend:
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8"}'
block-size: 128
kv-cache-dtype: fp8
gpu-memory-utilization: 0.95
max-model-len: 9472
language-model-only: true
Expand All @@ -66,8 +67,9 @@ backend:
trust-remote-code: true
no-enable-prefix-caching: true
kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8"}'
attention-config: '{"backend": "FLASHINFER", "use_trtllm_attention": true, "indexer_kv_dtype": "fp8", "minimax_m3_msa_decode_backend": "cutlass"}'
block-size: 128
kv-cache-dtype: fp8
gpu-memory-utilization: 0.95
max-model-len: 9472
language-model-only: true
Expand Down
2 changes: 1 addition & 1 deletion configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -7784,7 +7784,7 @@ minimaxm3-fp4-b300-dynamo-vllm-8k1k-legacy-max-tput:
dp-attn: false

minimaxm3-fp4-b300-dynamo-vllm-mtp:
image: vllm/vllm-openai:nightly-4080263bb2c5d10deac17aaeb88e0823bc35bca9
image: vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7
model: nvidia/MiniMax-M3-NVFP4
model-prefix: minimaxm3
runner: b300
Expand Down
7 changes: 7 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -5389,3 +5389,10 @@
- "Bump vLLM nightly image to nightly-5e35a6f4f9bbc217c599692157ca985c894373f7 (fixed-len MiniMax-M3 fix); update B300 FP4 8k1k tp1 recipe YAML container fields to match"
- "Enable VLLM_MINIMAX_M3_MSA_DECODE_BACKEND=cutlass in prefill and decode environments"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2473

- config-keys:
- minimaxm3-fp4-b300-dynamo-vllm-mtp
description:
- "Bump vLLM nightly image to nightly-5e35a6f4f9bbc217c599692157ca985c894373f7 (fixed-len MiniMax-M3 fix); update B300 FP4 MTP recipe YAML container fields to match (excluding 2p1d-dep2-dep4 used by legacy-dep4)"
- "Enable VLLM_MINIMAX_M3_MSA_DECODE_BACKEND=cutlass in prefill and decode environments"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2471