diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/b300-fp4-mtp/agentic.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/b300-fp4-mtp/agentic.yaml index 34c77fe338..05bf3f6020 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/b300-fp4-mtp/agentic.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/b300-fp4-mtp/agentic.yaml @@ -6,7 +6,7 @@ base: name: dsv41flash-fp4-b300-vllm-agentic model: path: hf:deepseek-ai/DeepSeek-V4.1-Flash - container: vllm/vllm-openai:deepseekv41-flash-0909 + container: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4 precision: fp4 resources: gpu_type: b300 @@ -24,7 +24,7 @@ base: # The engine may take the full VLLM_ENGINE_READY_TIMEOUT_S to become ready. health_check: interval_seconds: 10 - max_attempts: 360 + max_attempts: 720 roles: agg: nodes: 1 @@ -37,6 +37,10 @@ base: enable-auto-tool-choice: true reasoning-parser: deepseek_v41 engram-config: '{"cpu_offload":true}' + # FlashInfer sparse attention with the merged Blackwell sparse indexer + # (MXFP4 indexer KV, sparse logits) at both TP sizes; fp8 KV cache. + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + kv-cache-dtype: fp8 # Five-token DSpark with probabilistic drafting. Throughput runs replace # block rejection with the golden acceptance length. speculative-config: '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"probabilistic","rejection_sample_method":"block","enable_adaptive_verification":false}' @@ -46,7 +50,7 @@ base: env: VLLM_USE_RUST_FRONTEND: '1' PYTHONUNBUFFERED: '1' - VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_ENGINE_READY_TIMEOUT_S: '7200' benchmark: type: custom command: bash /infmax-workspace/benchmarks/srt_agentic.sh diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index b0fc9ef451..feace66ce7 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8328,7 +8328,7 @@ dsv41flash-fp4-gb200-sglang-agentic-dspark: - { tp: 2, ep: 1, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb200-fp4-mtp/agentic.yaml } dsv41flash-fp4-b300-vllm-agentic-dspark: - image: vllm/vllm-openai:deepseekv41-flash-0909 + image: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4 model: deepseek-ai/DeepSeek-V4.1-Flash model-prefix: dsv41flash runner: cluster:b300-dsxe diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 352c61fc0d..80e75c09b2 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8986,3 +8986,9 @@ - "Hopper 使用 FA3,Blackwell FA4 fp8 descale 问题不适用,EAGLE3 draft 保持 attention_backend FLASH_ATTN。" - "This change does not alter the EAGLE3 draft model data type. The draft loads unmodified from the published Inferact/MiniMax-M3-EAGLE3-GQA checkpoint via --speculative-config (method=eagle3). kv-cache-dtype fp8 sets KV-cache storage precision, not the draft weights, and no flag overrides or re-quantizes the draft weights." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3503 + +- config-keys: + - dsv41flash-fp4-b300-vllm-agentic-dspark + description: + - "Pin B300 DeepSeek-V4.1-Flash vLLM to nightly ddd6fbca with FlashInfer sparse attention, MXFP4 indexer KV, sparse logits, fp8 KV cache, and a 7200 s readiness timeout." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3458