diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml index 1b5e2787e2..d2ae0f834d 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml @@ -6,7 +6,7 @@ base: name: dsv41flash-fp4-gb300-vllm-agentic model: path: hf:deepseek-ai/DeepSeek-V4.1-Flash - container: vllm/vllm-openai:nightly-af1c01499b289be555c475669ba50a88e96d846e + container: vllm/vllm-openai:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d precision: fp4 resources: gpu_type: gb300 @@ -37,10 +37,14 @@ base: enable-auto-tool-choice: true reasoning-parser: deepseek_v41 engram-config: '{"cpu_offload":true}' + # FlashInfer sparse attention with MXFP4 indexer KV and sparse logits. + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + kv-cache-dtype: fp8 # Five-token DSpark with probabilistic drafting. Throughput runs replace # block rejection with the golden acceptance length. speculative-config: '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"probabilistic","rejection_sample_method":"block","enable_adaptive_verification":false}' max-model-len: 1048576 + max-num-seqs: 256 disable-uvicorn-access-log: true env: VLLM_USE_RUST_FRONTEND: '1' @@ -53,18 +57,20 @@ base: MODEL: deepseek-ai/DeepSeek-V4.1-Flash AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' -# One variant per point. Graph capture starts at 64 tokens and doubles until it -# covers CONC x (1 + 5 drafts), up to 2048. TP2 leaves ~175 GiB of weights on -# each 277 GiB GPU, so it caps batched tokens at 4096 (the indexer's 1M-wide -# logits buffer), bounds the scheduler at 2x CONC within [16, 256] (FlashInfer -# autotune fails below 16) and stops capturing above 512 tokens. +# One variant per point. Piecewise graph capture sizes are multiples of the +# six-token DSpark verification block, denser for small batches, and each +# batched-token limit matches the largest captured graph: CONC <= 4 and TP2 +# CONC 128 capture up to 2046 tokens (the latter at 0.97 memory utilization), +# every other point up to 8190. override_tp4_c1: roles: agg: gpus: 4 args: tensor-parallel-size: 4 - max-cudagraph-capture-size: 64 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}' + max-cudagraph-capture-size: 2046 + max-num-batched-tokens: 2048 benchmark: env: CONC: '1' @@ -75,7 +81,9 @@ override_tp4_c2: gpus: 4 args: tensor-parallel-size: 4 - max-cudagraph-capture-size: 64 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}' + max-cudagraph-capture-size: 2046 + max-num-batched-tokens: 2048 benchmark: env: CONC: '2' @@ -86,7 +94,9 @@ override_tp4_c4: gpus: 4 args: tensor-parallel-size: 4 - max-cudagraph-capture-size: 64 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}' + max-cudagraph-capture-size: 2046 + max-num-batched-tokens: 2048 benchmark: env: CONC: '4' @@ -97,7 +107,9 @@ override_tp4_c8: gpus: 4 args: tensor-parallel-size: 4 - max-cudagraph-capture-size: 64 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + max-cudagraph-capture-size: 8190 + max-num-batched-tokens: 8192 benchmark: env: CONC: '8' @@ -108,7 +120,9 @@ override_tp4_c16: gpus: 4 args: tensor-parallel-size: 4 - max-cudagraph-capture-size: 128 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + max-cudagraph-capture-size: 8190 + max-num-batched-tokens: 8192 benchmark: env: CONC: '16' @@ -119,7 +133,9 @@ override_tp4_c32: gpus: 4 args: tensor-parallel-size: 4 - max-cudagraph-capture-size: 256 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + max-cudagraph-capture-size: 8190 + max-num-batched-tokens: 8192 benchmark: env: CONC: '32' @@ -130,7 +146,9 @@ override_tp4_c64: gpus: 4 args: tensor-parallel-size: 4 - max-cudagraph-capture-size: 512 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + max-cudagraph-capture-size: 8190 + max-num-batched-tokens: 8192 benchmark: env: CONC: '64' @@ -141,33 +159,22 @@ override_tp4_c128: gpus: 4 args: tensor-parallel-size: 4 - max-cudagraph-capture-size: 1024 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + max-cudagraph-capture-size: 8190 + max-num-batched-tokens: 8192 benchmark: env: CONC: '128' -override_tp2_c1: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - max-cudagraph-capture-size: 64 - max-num-batched-tokens: 4096 - max-num-seqs: 16 - benchmark: - env: - CONC: '1' - override_tp2_c2: roles: agg: gpus: 2 args: tensor-parallel-size: 2 - max-cudagraph-capture-size: 64 - max-num-batched-tokens: 4096 - max-num-seqs: 16 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}' + max-cudagraph-capture-size: 2046 + max-num-batched-tokens: 2048 benchmark: env: CONC: '2' @@ -178,9 +185,9 @@ override_tp2_c4: gpus: 2 args: tensor-parallel-size: 2 - max-cudagraph-capture-size: 64 - max-num-batched-tokens: 4096 - max-num-seqs: 16 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}' + max-cudagraph-capture-size: 2046 + max-num-batched-tokens: 2048 benchmark: env: CONC: '4' @@ -191,9 +198,9 @@ override_tp2_c8: gpus: 2 args: tensor-parallel-size: 2 - max-cudagraph-capture-size: 64 - max-num-batched-tokens: 4096 - max-num-seqs: 16 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + max-cudagraph-capture-size: 8190 + max-num-batched-tokens: 8192 benchmark: env: CONC: '8' @@ -204,9 +211,9 @@ override_tp2_c16: gpus: 2 args: tensor-parallel-size: 2 - max-cudagraph-capture-size: 128 - max-num-batched-tokens: 4096 - max-num-seqs: 32 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + max-cudagraph-capture-size: 8190 + max-num-batched-tokens: 8192 benchmark: env: CONC: '16' @@ -217,9 +224,9 @@ override_tp2_c32: gpus: 2 args: tensor-parallel-size: 2 - max-cudagraph-capture-size: 256 - max-num-batched-tokens: 4096 - max-num-seqs: 64 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + max-cudagraph-capture-size: 8190 + max-num-batched-tokens: 8192 benchmark: env: CONC: '32' @@ -230,9 +237,9 @@ override_tp2_c64: gpus: 2 args: tensor-parallel-size: 2 - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 4096 - max-num-seqs: 128 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + max-cudagraph-capture-size: 8190 + max-num-batched-tokens: 8192 benchmark: env: CONC: '64' @@ -243,9 +250,23 @@ override_tp2_c128: gpus: 2 args: tensor-parallel-size: 2 - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 4096 - max-num-seqs: 256 + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}' + max-cudagraph-capture-size: 2046 + max-num-batched-tokens: 2048 + gpu-memory-utilization: 0.97 benchmark: env: CONC: '128' + +override_tp2_c1: + roles: + agg: + gpus: 2 + args: + tensor-parallel-size: 2 + max-cudagraph-capture-size: 64 + max-num-batched-tokens: 4096 + max-num-seqs: 16 + benchmark: + env: + CONC: '1' diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 07c47e6e40..aa436b20f3 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8396,7 +8396,7 @@ dsv41flash-fp4-b200-sglang-agentic-dspark: - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/agentic.yaml } dsv41flash-fp4-gb300-vllm-agentic-dspark: - image: vllm/vllm-openai:nightly-af1c01499b289be555c475669ba50a88e96d846e + image: vllm/vllm-openai:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d model: deepseek-ai/DeepSeek-V4.1-Flash model-prefix: dsv41flash runner: cluster:gb300-nv diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 0c952a9d34..ba06f7a359 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9139,3 +9139,9 @@ - "Drop TP4 concurrency 32." - "No data-type or precision change to the EAGLE3 draft model (Inferact/MiniMax-M3-EAGLE3-GQA): online_quant_config (ptpc_fp8, which ATOM also applies to the draft), kv_cache_dtype fp8 and index-cache-dtype fp8 are unchanged; the FlyDSL, mono-decode and lmcache_offload toggles affect the decode path and KV offload, not draft precision." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3388 + +- config-keys: + - dsv41flash-fp4-gb300-vllm-agentic-dspark + description: + - "Pin GB300 DeepSeek-V4.1-Flash vLLM to nightly ac68c308 with FlashInfer sparse attention, MXFP4 indexer KV, sparse logits, fp8 KV cache, and B300's batching and CUDA graph tiers." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3396