From 32a9df244c431475596c6b1323f3776460bf46a9 Mon Sep 17 00:00:00 2001 From: Xin Li Date: Sat, 1 Aug 2026 08:05:12 -0400 Subject: [PATCH 1/3] kimik2.6 b200 disagg: remove no-enable-flashinfer-autotune, runner b200-new --- .../b200-fp4/8k1k/disagg-b200-1p1d-dep4-dep8-c1024.yaml | 2 -- .../b200-fp4/8k1k/disagg-b200-1p1d-dep8-dep8-c2048.yaml | 2 -- .../b200-fp4/8k1k/disagg-b200-1p1d-dep8-tp8-c1.yaml | 2 -- .../b200-fp4/8k1k/disagg-b200-1p4d-dep4-tp4-c512.yaml | 2 -- .../b200-fp4/8k1k/disagg-b200-1p8d-dep4-tp4-c128.yaml | 2 -- .../b200-fp4/8k1k/disagg-b200-1p8d-dep4-tp4-c32.yaml | 2 -- .../b200-fp4/8k1k/disagg-b200-2p1d-dep8-dep8-c8192.yaml | 2 -- configs/nvidia-master.yaml | 2 +- perf-changelog.yaml | 6 ++++++ 9 files changed, 7 insertions(+), 15 deletions(-) diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p1d-dep4-dep8-c1024.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p1d-dep4-dep8-c1024.yaml index 5cb6c798b6..c639a4c102 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p1d-dep4-dep8-c1024.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p1d-dep4-dep8-c1024.yaml @@ -75,7 +75,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true attention-backend: FLASHINFER_MLA block-size: 128 attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED"}' @@ -95,7 +94,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true async-scheduling: true attention-backend: FLASHINFER_MLA block-size: 128 diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p1d-dep8-dep8-c2048.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p1d-dep8-dep8-c2048.yaml index e3c08dc477..8e100e0719 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p1d-dep8-dep8-c2048.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p1d-dep8-dep8-c2048.yaml @@ -75,7 +75,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true attention-backend: FLASHINFER_MLA block-size: 128 attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED"}' @@ -95,7 +94,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true async-scheduling: true attention-backend: FLASHINFER_MLA block-size: 128 diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p1d-dep8-tp8-c1.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p1d-dep8-tp8-c1.yaml index 1c0abe2fa0..31973e9d4e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p1d-dep8-tp8-c1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p1d-dep8-tp8-c1.yaml @@ -75,7 +75,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true attention-backend: FLASHINFER_MLA block-size: 128 attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED"}' @@ -93,7 +92,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true async-scheduling: true attention-backend: FLASHINFER_MLA block-size: 128 diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p4d-dep4-tp4-c512.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p4d-dep4-tp4-c512.yaml index 1ca87ad9b8..1c029057b7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p4d-dep4-tp4-c512.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p4d-dep4-tp4-c512.yaml @@ -75,7 +75,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true attention-backend: FLASHINFER_MLA block-size: 128 attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED"}' @@ -92,7 +91,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true async-scheduling: true attention-backend: FLASHINFER_MLA block-size: 128 diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p8d-dep4-tp4-c128.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p8d-dep4-tp4-c128.yaml index 44f8ff7849..040726623a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p8d-dep4-tp4-c128.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p8d-dep4-tp4-c128.yaml @@ -75,7 +75,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true attention-backend: FLASHINFER_MLA block-size: 128 attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED"}' @@ -92,7 +91,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true async-scheduling: true attention-backend: FLASHINFER_MLA block-size: 128 diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p8d-dep4-tp4-c32.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p8d-dep4-tp4-c32.yaml index c748107fec..5c8f046b46 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p8d-dep4-tp4-c32.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-1p8d-dep4-tp4-c32.yaml @@ -75,7 +75,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true attention-backend: FLASHINFER_MLA block-size: 128 attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED"}' @@ -92,7 +91,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true async-scheduling: true attention-backend: FLASHINFER_MLA block-size: 128 diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-2p1d-dep8-dep8-c8192.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-2p1d-dep8-dep8-c8192.yaml index 54e623d767..e7791715f6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-2p1d-dep8-dep8-c8192.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.6/b200-fp4/8k1k/disagg-b200-2p1d-dep8-dep8-c8192.yaml @@ -75,7 +75,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true attention-backend: FLASHINFER_MLA block-size: 128 attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED"}' @@ -95,7 +94,6 @@ backend: safetensors-load-strategy: prefetch trust-remote-code: true no-enable-prefix-caching: true - no-enable-flashinfer-autotune: true async-scheduling: true attention-backend: FLASHINFER_MLA block-size: 128 diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index a30af59641..449c7cc615 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -4981,7 +4981,7 @@ kimik2.6-fp4-b200-dynamo-vllm: image: vllm/vllm-openai:v0.25.1 model: nvidia/Kimi-K2.6-NVFP4 model-prefix: kimik2.6 - runner: b200-multinode + runner: b200-new precision: fp4 framework: dynamo-vllm router: { name: dynamo-router, version: "1.3.0.dev20260721" } diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 6eebd97070..98ca236263 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5355,3 +5355,9 @@ - "Apply the accuracy-gated Kimi-K2.5 MXFP4 settings: tuned AITER MXFP4 MoE, fused shared experts, FP8 KV cache, block size 16, 16384 batched tokens, 512 sequences, async scheduling, gpu-memory-utilization 0.85 (headroom for CUDA-graph capture on MI355X), and the AITER BF16 GEMM path" - "Extend the TP4 and TP8 8k1k concurrency sweep from 64 to 128 (1k1k deprecated per #2263)" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2213 + +- config-keys: + - kimik2.6-fp4-b200-dynamo-vllm + description: + - "Remove no-enable-flashinfer-autotune from Kimi K2.6 B200 disagg recipes, target runner b200-new" + pr-link: TBD From 92e4f9103a304a74ee7b70f41cffee5223bcb0a7 Mon Sep 17 00:00:00 2001 From: Xin Li Date: Sat, 1 Aug 2026 08:05:59 -0400 Subject: [PATCH 2/3] fill pr-link --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 98ca236263..a0c16a89d6 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5360,4 +5360,4 @@ - kimik2.6-fp4-b200-dynamo-vllm description: - "Remove no-enable-flashinfer-autotune from Kimi K2.6 B200 disagg recipes, target runner b200-new" - pr-link: TBD + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2445 From 67241fd83359be3468309b37de344363ce784754 Mon Sep 17 00:00:00 2001 From: Xin Li Date: Sat, 1 Aug 2026 08:16:48 -0400 Subject: [PATCH 3/3] revert runner to b200-multinode --- configs/nvidia-master.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 449c7cc615..a30af59641 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -4981,7 +4981,7 @@ kimik2.6-fp4-b200-dynamo-vllm: image: vllm/vllm-openai:v0.25.1 model: nvidia/Kimi-K2.6-NVFP4 model-prefix: kimik2.6 - runner: b200-new + runner: b200-multinode precision: fp4 framework: dynamo-vllm router: { name: dynamo-router, version: "1.3.0.dev20260721" }