From ca8c4ae8a08b00e17bf2b3bf4256691219c6704f Mon Sep 17 00:00:00 2001 From: Yuwei An Date: Thu, 24 Sep 2026 21:31:28 -0700 Subject: [PATCH 1/5] perf: update B300 DSV4 performance MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 更新 B300 DSV4 性能。 --- .../agentic/dsv4_fp4_b300_sglang_mtp.sh | 52 +++++++++++++++---- configs/nvidia-master.yaml | 14 +++++ perf-changelog.yaml | 9 ++++ 3 files changed, 65 insertions(+), 10 deletions(-) diff --git a/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh index 7adf2c573b..966274c8b0 100755 --- a/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh @@ -2,7 +2,7 @@ set -eo pipefail set -x -# DeepSeek-V4-Pro-0813 FP4 on B300 with SGLang DSpark K=6. +# DeepSeek-V4-Pro FP4 with EAGLE, or Pro-0813 with DSpark K=6, on B300. # KV_OFFLOADING=dram requires KV_OFFLOAD_BACKEND=hicache. SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -68,6 +68,9 @@ if require_agentic_kv_offload_backend hicache; then else HICACHE_RATIO=8 fi + if [ "$MODEL" = "deepseek-ai/DeepSeek-V4-Pro" ]; then + HICACHE_RATIO=2 + fi HICACHE_WRITE_POLICY="write_back" HICACHE_IO_BACKEND="direct" HICACHE_MEM_LAYOUT="page_first_direct" @@ -81,12 +84,13 @@ if require_agentic_kv_offload_backend hicache; then # AIPerf owns the AgentX warmup; SGLang's per-DP warmup can time out after # the API is already healthy. WARMUP_ARGS=(--skip-server-warmup) - echo "HiCache DSv4 CPU tier: ratio=$HICACHE_RATIO, capacity=${TOTAL_CPU_DRAM_GB} GB, write_policy=$HICACHE_WRITE_POLICY, io_backend=$HICACHE_IO_BACKEND, mem_layout=$HICACHE_MEM_LAYOUT" + echo "HiCache DSv4 CPU tier: ratio=$HICACHE_RATIO, cpu_budget=${TOTAL_CPU_DRAM_GB} GB, write_policy=$HICACHE_WRITE_POLICY, io_backend=$HICACHE_IO_BACKEND, mem_layout=$HICACHE_MEM_LAYOUT" fi USE_SGLANG_ROUTER=false SGLANG_BACKEND_PORT="$PORT" ROUTER_LOG="$RESULT_DIR/router.log" +ROUTER_ARGS=() if [ "$DP_ATTENTION" = "true" ]; then USE_SGLANG_ROUTER=true export AIPERF_HTTP_X_SMG_ROUTING_KEY_FROM_CORRELATION_ID=true @@ -100,13 +104,16 @@ METRICS_ARGS=(--enable-metrics --enable-cache-report) MEM_FRACTION_STATIC=0.88 CHUNKED_PREFILL_SIZE=8192 if [ "$DP_ATTENTION" = "true" ]; then + PREFILL_DECODE_INTERVAL=20 + if [ "$MODEL" = "deepseek-ai/DeepSeek-V4-Pro" ]; then + PREFILL_DECODE_INTERVAL=32 + fi PARALLEL_ARGS+=( --dp "$TP" --tokenizer-worker-num "$TP" --enable-prefill-delayer - --prefill-decode-interval 20 + --prefill-decode-interval "$PREFILL_DECODE_INTERVAL" --enable-dp-attention - --enable-dp-lm-head --enable-dp-attention-local-control-broadcast --incremental-streaming-output --stream-interval 20 @@ -140,6 +147,14 @@ if [ "$DP_ATTENTION" = "true" ]; then # Scale it so every DEP shape gets 8192 per rank; 16384/rank exceeds # MegaMoE's per-rank token cap (startup ValueError). CHUNKED_PREFILL_SIZE=$((8192 * TP)) + if [ "$MODEL" = "deepseek-ai/DeepSeek-V4-Pro" ]; then + MEM_FRACTION_STATIC=0.84 + PARALLEL_ARGS+=(--enable-mixed-chunk --schedule-policy shortest-prefill-first) + export SGLANG_ENABLE_DP_SPEC_PREFILL_COORDINATION=1 + ROUTER_ARGS+=(--disable-circuit-breaker) + else + PARALLEL_ARGS+=(--enable-dp-lm-head) + fi else PARALLEL_ARGS+=( --moe-runner-backend flashinfer_mxfp4 @@ -152,6 +167,9 @@ MODEL_ARGS=( --page-size 256 --disable-shared-experts-fusion ) +if [ "$MODEL" = "deepseek-ai/DeepSeek-V4-Pro" ]; then + MODEL_ARGS=(--attention-backend dsv4 --page-size 256 --disable-shared-experts-fusion) +fi # AgentX concurrency counts live session trees, not individual requests. # Allow subagent fan-out to exceed CONC without clipping request bursts. @@ -192,8 +210,25 @@ if [ "$DP_ATTENTION" = "true" ]; then # extra 128 is headroom over the exact-fit boundary. export SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320 fi +SPEC_ARGS=( + --speculative-algorithm DSPARK + --speculative-dspark-block-size 6 + --speculative-num-steps 1 + --speculative-eagle-topk 1 + --speculative-num-draft-tokens 7 +) +SYNTHETIC_ACC_LEN=3.77 +if [ "$MODEL" = "deepseek-ai/DeepSeek-V4-Pro" ]; then + SPEC_ARGS=( + --speculative-algorithm EAGLE + --speculative-num-steps 3 + --speculative-eagle-topk 1 + --speculative-num-draft-tokens 4 + ) + SYNTHETIC_ACC_LEN=2.49 +fi if [ "${EVAL_ONLY}" != "true" ]; then - export SGLANG_SIMULATE_ACC_LEN=3.77 + export SGLANG_SIMULATE_ACC_LEN="$SYNTHETIC_ACC_LEN" export SGLANG_SIMULATE_ACC_METHOD=match-expected export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token fi @@ -224,11 +259,7 @@ SGLANG_CMD=( --reasoning-parser deepseek-v4 --chat-template "$SCRIPT_DIR/../chat_templates/deepseek_v4_thinking.jinja" --watchdog-timeout 1800 - --speculative-algorithm DSPARK - --speculative-dspark-block-size 6 - --speculative-num-steps 1 - --speculative-eagle-topk 1 - --speculative-num-draft-tokens 7 + "${SPEC_ARGS[@]}" "${MODEL_ARGS[@]}" "${METRICS_ARGS[@]}" "${CACHE_ARGS[@]}" @@ -277,6 +308,7 @@ if [ "$USE_SGLANG_ROUTER" = "true" ]; then --connect-timeout-secs 900 \ --request-timeout-secs 14400 \ --disable-health-check \ + "${ROUTER_ARGS[@]}" \ `# A single transient router->engine send failure would otherwise` \ `# surface as a 500, and AgentX aborts the whole run when a root` \ `# warmup request fails ("ProfileAborted"). Measured at conc 512:` \ diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index e38ba4bc9e..69865ed9c6 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -1145,6 +1145,20 @@ dsv4-fp4-b300-sglang-agentic-hicache-mtp: # Both paths share DSpark K=6 (1 step, 7 draft tokens) and # max-running-requests 2*CONC. +dsv4-fp4-b300-sglang-agentic-hicache-eagle: + image: lmsysorg/sglang:dev-cu13-nightly-09242@sha256:c47bbe7448050608b4e08669c742df4835a45a95bf6b19f2c4e51a7f6609e9d5 + model: deepseek-ai/DeepSeek-V4-Pro + model-prefix: dsv4 + runner: cluster:b300-dsxe + precision: fp4 + framework: sglang + multinode: false + scenarios: + agentic-coding: + - dram-utilization: 0.95 + search-space: + - { tp: 8, ep: 8, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: hicache }, spec-decoding: mtp, conc-list: [64, 128, 256, 384, 512], router: { name: sglang-router, version: "0.3.2" } } + qwen3.5-fp8-b200-sglang: image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 model: Qwen/Qwen3.5-397B-A17B-FP8 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 1bc132cd37..1d61229ce1 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8950,3 +8950,12 @@ description: - "Update SGLang image from v0.5.19-cu130 to v0.5.20-cu130 and pass --cuda-graph-max-bs-decode after the alias removal." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3334 + +- config-keys: + - dsv4-fp4-b300-sglang-agentic-hicache-eagle + scenario-type: + - agentic-coding + description: + - "Update B300 DSV4 performance." + - "更新 B300 DSV4 性能。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX From d47e631d967c25a300439778c5690bb44d95ae9f Mon Sep 17 00:00:00 2001 From: Yuwei An Date: Thu, 24 Sep 2026 22:17:13 -0700 Subject: [PATCH 2/5] perf: include B300 DSV4 low-latency points MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 补充 B300 DSV4 TP8 低延迟并发点 1、2、4、8、16、32。 --- configs/nvidia-master.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 69865ed9c6..a4ad60f7dd 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -1157,6 +1157,7 @@ dsv4-fp4-b300-sglang-agentic-hicache-eagle: agentic-coding: - dram-utilization: 0.95 search-space: + - { tp: 8, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32] } - { tp: 8, ep: 8, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: hicache }, spec-decoding: mtp, conc-list: [64, 128, 256, 384, 512], router: { name: sglang-router, version: "0.3.2" } } qwen3.5-fp8-b200-sglang: From 86882167bf00f5e7e72e71a8930a340d9dd06df9 Mon Sep 17 00:00:00 2001 From: Yuwei An Date: Thu, 24 Sep 2026 22:26:47 -0700 Subject: [PATCH 3/5] fix: use explicit decode graph flag for DSV4 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit DSV4 低延迟配置使用明确的 decode CUDA graph 参数,修复新镜像的 CLI 歧义错误。 --- benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh index 966274c8b0..12e48370fb 100755 --- a/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh @@ -179,9 +179,10 @@ MAX_RUNNING_REQUESTS=$((2 * CONC)) CUDA_GRAPH_MAX_BS=$((CONC * 4)) [ "$CUDA_GRAPH_MAX_BS" -gt 64 ] && CUDA_GRAPH_MAX_BS=64 -# --cuda-graph-max-bs is an alias whose dest is cuda_graph_max_bs_decode, so the -# two forms below are the same knob and must not both be passed. CUDA_GRAPH_ARGS=(--cuda-graph-max-bs "$CUDA_GRAPH_MAX_BS") +if [ "$MODEL" = "deepseek-ai/DeepSeek-V4-Pro" ]; then + CUDA_GRAPH_ARGS=(--cuda-graph-max-bs-decode "$CUDA_GRAPH_MAX_BS") +fi SWA_FULL_TOKENS_RATIO=0.1 if [ "$DP_ATTENTION" = "true" ]; then # Decode graphs must cover the padded speculative batch across all DP ranks, which From 2e46f34fe78627d0b0b108301dbacca77f360062 Mon Sep 17 00:00:00 2001 From: Yuwei An Date: Fri, 25 Sep 2026 13:27:20 -0700 Subject: [PATCH 4/5] perf: add B300 DSV4 TP4 low-latency points MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 添加 B300 DSV4 TP4 低延迟测试点,并保留现有 TP8 和 DP8 配置。 --- configs/nvidia-master.yaml | 1 + perf-changelog.yaml | 9 +++++++++ 2 files changed, 10 insertions(+) diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index a4ad60f7dd..8aa31b2cae 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -1157,6 +1157,7 @@ dsv4-fp4-b300-sglang-agentic-hicache-eagle: agentic-coding: - dram-utilization: 0.95 search-space: + - { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32] } - { tp: 8, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32] } - { tp: 8, ep: 8, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: hicache }, spec-decoding: mtp, conc-list: [64, 128, 256, 384, 512], router: { name: sglang-router, version: "0.3.2" } } diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 1d61229ce1..ab63defc92 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8959,3 +8959,12 @@ - "Update B300 DSV4 performance." - "更新 B300 DSV4 性能。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + +- config-keys: + - dsv4-fp4-b300-sglang-agentic-hicache-eagle + scenario-type: + - agentic-coding + description: + - "Add B300 DSV4 TP4 low-latency points." + - "添加 B300 DSV4 TP4 低延迟测试点。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3426 From f9bdb7683235c65624ba1100a1f866490e52c8d1 Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Mon, 28 Sep 2026 13:12:59 -0500 Subject: [PATCH 5/5] perf(b300): port DSV4 SGLang EAGLE AgentX to a native SRT recipe --- .../dsv4/sglang/b300-fp4-eagle/agentic.yaml | 335 ++++++++++++++++++ inferencex-e2e/configs/nvidia-master.yaml | 6 +- inferencex-e2e/perf-changelog.yaml | 10 + 3 files changed, 348 insertions(+), 3 deletions(-) create mode 100644 inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml new file mode 100644 index 0000000000..bdcd7c61de --- /dev/null +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml @@ -0,0 +1,335 @@ +# DeepSeek-V4-Pro AgentX on B300 with SGLang EAGLE (3 steps, 4 draft tokens). +# TP4 and TP8 run the flashinfer MXFP4 MoE with GPU-resident KV; DEP8 +# (attention DP + Mega-MoE + FP4 indexer) runs behind the SGLang router with a +# HiCache DRAM tier. +base: + schema: 2 + name: dsv4-fp4-b300-sglang-agentic-eagle + model: + path: hf:deepseek-ai/DeepSeek-V4-Pro + container: lmsysorg/sglang:dev-cu13-nightly-09242@sha256:c47bbe7448050608b4e08669c742df4835a45a95bf6b19f2c4e51a7f6609e9d5 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro + trust-remote-code: true + tensor-parallel-size: 8 + moe-runner-backend: flashinfer_mxfp4 + disable-flashinfer-autotune: true + mem-fraction-static: 0.88 + swa-full-tokens-ratio: 0.1 + allow-auto-truncate: true + chunked-prefill-size: 8192 + tool-call-parser: deepseekv4 + reasoning-parser: deepseek-v4 + chat-template: /infmax-workspace/benchmarks/single_node/chat_templates/deepseek_v4_thinking.jinja + watchdog-timeout: 1800 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + attention-backend: dsv4 + page-size: 256 + disable-shared-experts-fusion: true + enable-metrics: true + enable-cache-report: true + env: + SGLANG_ENABLE_UNIFIED_RADIX_TREE: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + PYTHONNOUSERSITE: '1' + TORCH_CUDA_ARCH_LIST: '10.0' + # Outlast AIPerf's pooled connections past Uvicorn's keep-alive. + SGLANG_TIMEOUT_KEEP_ALIVE: '900' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_USE_JIT_NORM: '1' + SGLANG_OPT_USE_JIT_INDEXER_METADATA: '1' + SGLANG_OPT_USE_TOPK_V2: '1' + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + # Triton compiles with the image's CUDA ptxas. + TRITON_PTXAS_PATH: /usr/local/cuda/bin/ptxas + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + MODEL: deepseek-ai/DeepSeek-V4-Pro + # Warmup can leave bytes unacknowledged past AIPerf's 30 s default. + AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' + +# One variant per point. TP4/TP8: admission is 2x CONC and the decode graph +# batch 4x CONC, capped at 64. +override_tp4_c1: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 2 + cuda-graph-max-bs-decode: 4 + benchmark: + env: + CONC: '1' + KV_OFFLOADING: none + +override_tp4_c2: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 4 + cuda-graph-max-bs-decode: 8 + benchmark: + env: + CONC: '2' + KV_OFFLOADING: none + +override_tp4_c4: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 8 + cuda-graph-max-bs-decode: 16 + benchmark: + env: + CONC: '4' + KV_OFFLOADING: none + +override_tp4_c8: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 16 + cuda-graph-max-bs-decode: 32 + benchmark: + env: + CONC: '8' + KV_OFFLOADING: none + +override_tp4_c16: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 32 + cuda-graph-max-bs-decode: 64 + benchmark: + env: + CONC: '16' + KV_OFFLOADING: none + +override_tp4_c32: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 64 + cuda-graph-max-bs-decode: 64 + benchmark: + env: + CONC: '32' + KV_OFFLOADING: none + +override_tp8_c1: + roles: + agg: + args: + max-running-requests: 2 + cuda-graph-max-bs-decode: 4 + benchmark: + env: + CONC: '1' + KV_OFFLOADING: none + +override_tp8_c2: + roles: + agg: + args: + max-running-requests: 4 + cuda-graph-max-bs-decode: 8 + benchmark: + env: + CONC: '2' + KV_OFFLOADING: none + +override_tp8_c4: + roles: + agg: + args: + max-running-requests: 8 + cuda-graph-max-bs-decode: 16 + benchmark: + env: + CONC: '4' + KV_OFFLOADING: none + +override_tp8_c8: + roles: + agg: + args: + max-running-requests: 16 + cuda-graph-max-bs-decode: 32 + benchmark: + env: + CONC: '8' + KV_OFFLOADING: none + +override_tp8_c16: + roles: + agg: + args: + max-running-requests: 32 + cuda-graph-max-bs-decode: 64 + benchmark: + env: + CONC: '16' + KV_OFFLOADING: none + +override_tp8_c32: + roles: + agg: + args: + max-running-requests: 64 + cuda-graph-max-bs-decode: 64 + benchmark: + env: + CONC: '32' + KV_OFFLOADING: none + +# DEP8 replaces the TP8 MoE path: the router keeps each session on the DP rank +# holding its radix prefix, the global prefill chunk is 8192 per rank, decode +# graphs cover the padded speculative batch across ranks, and AIPerf owns the +# warmup. DEP8 mixes prefill into decode batches, schedules shortest prefill +# first, coordinates speculative prefill across DP ranks, and keeps the router +# circuit breaker off. HiCache capacity is a host/device ratio of 2. +override_dep8_c64: + frontend: &dep8_frontend + type: sglang-router + args: + policy: consistent_hashing + request-id-headers: x-correlation-id + dp-aware: true + connect-timeout-secs: 900 + request-timeout-secs: 14400 + disable-health-check: true + disable-circuit-breaker: true + # A transient router-to-engine send failure would otherwise abort the run. + retry-max-retries: 8 + retry-initial-backoff-ms: 500 + retry-max-backoff-ms: 10000 + retry-backoff-multiplier: 2 + roles: + agg: + args: + <<: &dep8_args + moe-runner-backend: null + data-parallel-size: 8 + tokenizer-worker-num: 8 + enable-prefill-delayer: true + prefill-decode-interval: 32 + enable-dp-attention: true + enable-dp-attention-local-control-broadcast: true + incremental-streaming-output: true + stream-interval: 20 + expert-parallel-size: 8 + moe-a2a-backend: megamoe + enable-w4a4-mxfp4-megamoe: true + enable-deepseek-v4-fp4-indexer: true + enable-mixed-chunk: true + schedule-policy: shortest-prefill-first + mem-fraction-static: 0.84 + chunked-prefill-size: 65536 + cuda-graph-max-bs-decode: 544 + swa-full-tokens-ratio: 0.075 + enable-hierarchical-cache: true + hicache-ratio: 2 + hicache-write-policy: write_back + hicache-io-backend: direct + hicache-mem-layout: page_first_direct + skip-server-warmup: true + max-running-requests: 128 + env: &dep8_env + SGLANG_ENABLE_DP_SPEC_PREFILL_COORDINATION: '1' + # Covers the 8192-token per-rank prefill budget. + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '8320' + benchmark: + env: + <<: &dep8_client + AIPERF_HTTP_X_SMG_ROUTING_KEY_FROM_CORRELATION_ID: 'true' + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '2849' + CONC: '64' + +override_dep8_c128: + frontend: *dep8_frontend + roles: + agg: + args: + <<: *dep8_args + max-running-requests: 256 + env: *dep8_env + benchmark: + env: + <<: *dep8_client + CONC: '128' + +override_dep8_c256: + frontend: *dep8_frontend + roles: + agg: + args: + <<: *dep8_args + max-running-requests: 512 + env: *dep8_env + benchmark: + env: + <<: *dep8_client + CONC: '256' + +override_dep8_c384: + frontend: *dep8_frontend + roles: + agg: + args: + <<: *dep8_args + max-running-requests: 768 + env: *dep8_env + benchmark: + env: + <<: *dep8_client + CONC: '384' + +override_dep8_c512: + frontend: *dep8_frontend + roles: + agg: + args: + <<: *dep8_args + max-running-requests: 1024 + env: *dep8_env + benchmark: + env: + <<: *dep8_client + CONC: '512' diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index d3b51b158d..e10fea7925 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -1157,9 +1157,9 @@ dsv4-fp4-b300-sglang-agentic-hicache-eagle: agentic-coding: - dram-utilization: 0.95 search-space: - - { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32] } - - { tp: 8, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32] } - - { tp: 8, ep: 8, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: hicache }, spec-decoding: mtp, conc-list: [64, 128, 256, 384, 512], router: { name: sglang-router, version: "0.3.2" } } + - { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml } + - { tp: 8, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml } + - { tp: 8, ep: 8, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: hicache }, spec-decoding: mtp, conc-list: [64, 128, 256, 384, 512], router: { name: sglang-router, version: "0.3.2" }, srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml } qwen3.5-fp8-b200-sglang: image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index a0ecd2fc71..5e6cc3ac1b 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9003,3 +9003,13 @@ description: - "Update B300 vLLM AgentX to DSpark6 on a new image with a sampled concurrency grid and per-mode --kv-cache-memory-bytes pins." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3477 + +- config-keys: + - dsv4-fp4-b300-sglang-agentic-hicache-eagle + scenario-type: + - agentic-coding + description: + - "Add B300 DeepSeek-V4-Pro SGLang AgentX with EAGLE (3 steps, 4 draft tokens, golden AL 2.49) on lmsysorg/sglang:dev-cu13-nightly-09242: TP4 and TP8 at c1-32 with GPU KV, and DEP8 with a HiCache DRAM tier (ratio 2) at c64-512." + - "DEP8 uses the dsv4 attention backend, mem-fraction-static 0.84, prefill-decode-interval 32, mixed chunking, shortest-prefill-first scheduling, DP speculative prefill coordination and a router with the circuit breaker disabled, without the DP LM head." + - "Runs on the native srt-slurm recipe benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3426