Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ base:
name: dsv41flash-fp4-gb300-vllm-agentic
model:
path: hf:deepseek-ai/DeepSeek-V4.1-Flash
container: vllm/vllm-openai:nightly-af1c01499b289be555c475669ba50a88e96d846e
container: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4
precision: fp4
resources:
gpu_type: gb300
Expand Down Expand Up @@ -37,10 +37,14 @@ base:
enable-auto-tool-choice: true
reasoning-parser: deepseek_v41
engram-config: '{"cpu_offload":true}'
# FlashInfer sparse attention with MXFP4 indexer KV and sparse logits.
attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}'
kv-cache-dtype: fp8
# Five-token DSpark with probabilistic drafting. Throughput runs replace
# block rejection with the golden acceptance length.
speculative-config: '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"probabilistic","rejection_sample_method":"block","enable_adaptive_verification":false}'
max-model-len: 1048576
max-num-seqs: 256
disable-uvicorn-access-log: true
env:
VLLM_USE_RUST_FRONTEND: '1'
Expand All @@ -53,18 +57,20 @@ base:
MODEL: deepseek-ai/DeepSeek-V4.1-Flash
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:'

# One variant per point. Graph capture starts at 64 tokens and doubles until it
# covers CONC x (1 + 5 drafts), up to 2048. TP2 leaves ~175 GiB of weights on
# each 277 GiB GPU, so it caps batched tokens at 4096 (the indexer's 1M-wide
# logits buffer), bounds the scheduler at 2x CONC within [16, 256] (FlashInfer
# autotune fails below 16) and stops capturing above 512 tokens.
# One variant per point. Piecewise graph capture sizes are multiples of the
# six-token DSpark verification block, denser for small batches, and each
# batched-token limit matches the largest captured graph: CONC <= 4 and TP2
# CONC 128 capture up to 2046 tokens (the latter at 0.97 memory utilization),
# every other point up to 8190.
override_tp4_c1:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
max-cudagraph-capture-size: 2046
max-num-batched-tokens: 2048
benchmark:
env:
CONC: '1'
Expand All @@ -75,7 +81,9 @@ override_tp4_c2:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
max-cudagraph-capture-size: 2046
max-num-batched-tokens: 2048
benchmark:
env:
CONC: '2'
Expand All @@ -86,7 +94,9 @@ override_tp4_c4:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
max-cudagraph-capture-size: 2046
max-num-batched-tokens: 2048
benchmark:
env:
CONC: '4'
Expand All @@ -97,7 +107,9 @@ override_tp4_c8:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '8'
Expand All @@ -108,7 +120,9 @@ override_tp4_c16:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 128
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '16'
Expand All @@ -119,7 +133,9 @@ override_tp4_c32:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 256
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '32'
Expand All @@ -130,7 +146,9 @@ override_tp4_c64:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 512
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '64'
Expand All @@ -141,33 +159,22 @@ override_tp4_c128:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 1024
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '128'

override_tp2_c1:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
benchmark:
env:
CONC: '1'

override_tp2_c2:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
max-cudagraph-capture-size: 2046
max-num-batched-tokens: 2048
benchmark:
env:
CONC: '2'
Expand All @@ -178,9 +185,9 @@ override_tp2_c4:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
max-cudagraph-capture-size: 2046
max-num-batched-tokens: 2048
benchmark:
env:
CONC: '4'
Expand All @@ -191,9 +198,9 @@ override_tp2_c8:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '8'
Expand All @@ -204,9 +211,9 @@ override_tp2_c16:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 128
max-num-batched-tokens: 4096
max-num-seqs: 32
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '16'
Expand All @@ -217,9 +224,9 @@ override_tp2_c32:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 256
max-num-batched-tokens: 4096
max-num-seqs: 64
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '32'
Expand All @@ -230,9 +237,9 @@ override_tp2_c64:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 512
max-num-batched-tokens: 4096
max-num-seqs: 128
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '64'
Expand All @@ -243,9 +250,23 @@ override_tp2_c128:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 512
max-num-batched-tokens: 4096
max-num-seqs: 256
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
max-cudagraph-capture-size: 2046
max-num-batched-tokens: 2048
gpu-memory-utilization: 0.97
benchmark:
env:
CONC: '128'

override_tp2_c1:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
benchmark:
env:
CONC: '1'
2 changes: 1 addition & 1 deletion configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8386,7 +8386,7 @@ dsv41flash-fp4-b200-sglang-agentic-dspark:
- { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/agentic.yaml }

dsv41flash-fp4-gb300-vllm-agentic-dspark:
image: vllm/vllm-openai:nightly-af1c01499b289be555c475669ba50a88e96d846e
image: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4
model: deepseek-ai/DeepSeek-V4.1-Flash
model-prefix: dsv41flash
runner: cluster:gb300-nv
Expand Down
6 changes: 6 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8986,3 +8986,9 @@
- "Hopper 使用 FA3,Blackwell FA4 fp8 descale 问题不适用,EAGLE3 draft 保持 attention_backend FLASH_ATTN。"
- "This change does not alter the EAGLE3 draft model data type. The draft loads unmodified from the published Inferact/MiniMax-M3-EAGLE3-GQA checkpoint via --speculative-config (method=eagle3). kv-cache-dtype fp8 sets KV-cache storage precision, not the draft weights, and no flag overrides or re-quantizes the draft weights."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3503

- config-keys:
- dsv41flash-fp4-gb300-vllm-agentic-dspark
description:
- "Pin GB300 DeepSeek-V4.1-Flash vLLM to nightly ddd6fbca with FlashInfer sparse attention, MXFP4 indexer KV, sparse logits, fp8 KV cache, and B300's batching and CUDA graph tiers."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3396
Loading