From ba1869546212345121975eefa65c3aafcc99ee7c Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Sat, 26 Sep 2026 06:43:42 +0000 Subject: [PATCH] chore(recipes): remove unreferenced srt-slurm YAMLs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Remove 20 unused recipe files without changing active or deprecated config references. Generated matrices are unchanged.\n\n删除 20 个未引用的 srt-slurm 配方文件,不修改活动或已弃用配置的引用,生成矩阵保持不变。 --- .../8k1k/disagg-1p1d-dep4-dep8-c308-stp.yaml | 131 ----------- .../8k1k/disagg-1p4d-dep4-tep8-c24-stp.yaml | 127 ----------- .../8k1k/disagg-1p4d-dep4-tep8-c4-stp.yaml | 127 ----------- .../8k1k/disagg-1p5d-dep4-tep4-c115-stp.yaml | 129 ----------- .../8k1k/disagg-1p5d-dep4-tep4-c195-stp.yaml | 131 ----------- .../8k1k/disagg-1p5d-dep4-tep4-c30-stp.yaml | 127 ----------- .../8k1k/disagg-1p5d-dep4-tep4-c5-stp.yaml | 127 ----------- .../8k1k/disagg-1p5d-dep4-tep4-c60-stp.yaml | 128 ----------- .../8k1k/disagg-2p1d-dep4-dep8-c615-stp.yaml | 135 ------------ .../8k1k/disagg-3p1d-dep4-dep8-c1127-stp.yaml | 143 ------------ .../8k1k/disagg-4p1d-dep4-dep8-c2151-stp.yaml | 159 -------------- .../agentx/agg-dep4-vllm-simple.yaml | 108 ---------- .../vllm/gb200-fp4/agentx/agg-dep4.yaml | 106 --------- .../vllm/gb200-fp4/agentx/agg-dep8.yaml | 106 --------- .../gb200-fp4/agentx/agg-tp4-vllm-simple.yaml | 113 ---------- .../vllm/gb200-fp4/agentx/agg-tp4.yaml | 116 ---------- .../agentx/disagg-1p1d-dep8-dep4.yaml | 142 ------------ .../gb200-fp4/agentx/agg-tp4-mtp-hicache.yaml | 88 -------- .../kimik3/vllm/mi355x-fp4-mtp/agentic.yaml | 185 ---------------- .../vllm/mi355x-fp4-mtp/agentic.yaml | 204 ------------------ 20 files changed, 2632 deletions(-) delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p1d-dep4-dep8-c308-stp.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c24-stp.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c4-stp.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c115-stp.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c195-stp.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c30-stp.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c5-stp.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c60-stp.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-2p1d-dep4-dep8-c615-stp.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-3p1d-dep4-dep8-c1127-stp.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-4p1d-dep4-dep8-c2151-stp.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-dep4-vllm-simple.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-dep4.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-dep8.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-vllm-simple.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep4.yaml delete mode 100644 benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-hicache.yaml delete mode 100644 benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4-mtp/agentic.yaml delete mode 100644 benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi355x-fp4-mtp/agentic.yaml diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p1d-dep4-dep8-c308-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p1d-dep4-dep8-c308-stp.yaml deleted file mode 100644 index 7b70090991..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p1d-dep4-dep8-c308-stp.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: disagg-B200-1p1d-dep4-dep8-c308-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 1 - workers: 1 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.92 - max_batch_size: 32 - max_num_tokens: 32 - max_seq_len: 9256 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 308 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c24-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c24-stp.yaml deleted file mode 100644 index 9efed38c1d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c24-stp.yaml +++ /dev/null @@ -1,127 +0,0 @@ -schema: 2 -name: disagg-B200-1p4d-dep4-tep8-c24-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 4 - workers: 4 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 4 - max_num_tokens: 4 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 24 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c4-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c4-stp.yaml deleted file mode 100644 index 33e18fd915..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c4-stp.yaml +++ /dev/null @@ -1,127 +0,0 @@ -schema: 2 -name: disagg-B200-1p4d-dep4-tep8-c4-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 4 - workers: 4 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 1 - max_num_tokens: 1 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 4 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c115-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c115-stp.yaml deleted file mode 100644 index b03bbea144..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c115-stp.yaml +++ /dev/null @@ -1,129 +0,0 @@ -schema: 2 -name: disagg-B200-1p5d-dep4-tep4-c115-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 3 - workers: 5 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 16 - max_num_tokens: 16 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 4 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 115 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c195-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c195-stp.yaml deleted file mode 100644 index 82ddd6648c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c195-stp.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: disagg-B200-1p5d-dep4-tep4-c195-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 3 - workers: 5 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 32 - max_num_tokens: 32 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 4 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 195 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c30-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c30-stp.yaml deleted file mode 100644 index 939cefcefd..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c30-stp.yaml +++ /dev/null @@ -1,127 +0,0 @@ -schema: 2 -name: disagg-B200-1p5d-dep4-tep4-c30-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 3 - workers: 5 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 4 - max_num_tokens: 4 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 4 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 30 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c5-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c5-stp.yaml deleted file mode 100644 index 8638021360..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c5-stp.yaml +++ /dev/null @@ -1,127 +0,0 @@ -schema: 2 -name: disagg-B200-1p5d-dep4-tep4-c5-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 3 - workers: 5 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 1 - max_num_tokens: 1 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 4 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 5 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c60-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c60-stp.yaml deleted file mode 100644 index c5cc9ceab4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c60-stp.yaml +++ /dev/null @@ -1,128 +0,0 @@ -schema: 2 -name: disagg-B200-1p5d-dep4-tep4-c60-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 3 - workers: 5 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 8 - max_num_tokens: 8 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 4 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 60 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-2p1d-dep4-dep8-c615-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-2p1d-dep4-dep8-c615-stp.yaml deleted file mode 100644 index 028a790246..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-2p1d-dep4-dep8-c615-stp.yaml +++ /dev/null @@ -1,135 +0,0 @@ -schema: 2 -name: disagg-B200-2p1d-dep4-dep8-c615-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 2 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 1 - workers: 1 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.92 - max_batch_size: 64 - max_num_tokens: 64 - max_seq_len: 9256 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 615 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-3p1d-dep4-dep8-c1127-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-3p1d-dep4-dep8-c1127-stp.yaml deleted file mode 100644 index 503f6c83a8..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-3p1d-dep4-dep8-c1127-stp.yaml +++ /dev/null @@ -1,143 +0,0 @@ -schema: 2 -name: disagg-B200-3p1d-dep4-dep8-c1127-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 3 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 1 - workers: 1 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.92 - max_batch_size: 128 - max_num_tokens: 128 - max_seq_len: 9256 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 1127 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-4p1d-dep4-dep8-c2151-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-4p1d-dep4-dep8-c2151-stp.yaml deleted file mode 100644 index d7a6aad22c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-4p1d-dep4-dep8-c2151-stp.yaml +++ /dev/null @@ -1,159 +0,0 @@ -schema: 2 -name: disagg-B200-4p1d-dep4-dep8-c2151-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 4 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 1 - workers: 1 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - - 136 - - 144 - - 152 - - 160 - - 168 - - 176 - - 184 - - 192 - - 200 - - 208 - - 216 - - 224 - - 232 - - 240 - - 248 - - 256 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.92 - max_batch_size: 256 - max_num_tokens: 256 - max_seq_len: 9256 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 2151 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-dep4-vllm-simple.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-dep4-vllm-simple.yaml deleted file mode 100644 index 1ec16d4f66..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-dep4-vllm-simple.yaml +++ /dev/null @@ -1,108 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-agg-gb200-dep4-vllm-simple-agentic" - -model: {path: "minimax-m3-nvfp4", container: "vllm/vllm-openai:v0.27.1", precision: "fp4"} -identity: - model: {repo: "nvidia/MiniMax-M3-NVFP4"} - container: {image: "vllm/vllm-openai:v0.27.1"} - frameworks: {dynamo: "1.3.1"} -dynamo: {install: true, source: {pypi: "1.3.1"}} -environment: {ETCD_LEASE_TTL: "7200"} -slurm: {time_limit: "12:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} -resources: {gpu_type: "gb200", gpus_per_node: 4} -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - env: {DYN_TCP_REQUEST_TIMEOUT: "60"} - args: - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "minimax_m3" - reasoning-parser: "minimax_m3" - enable-auto-tool-choice: true - router-mode: "kv" - router-kv-events: true - router-reset-states: true - router-temperature: "0" - router-session-affinity-ttl-secs: 14400 - kv-cache-block-size: 128 -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - env: - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_USE_SIMPLE_KV_OFFLOAD: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "cuda_copy,cuda_ipc,rc" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 4 - data-parallel-rpc-port: 13345 - enable-expert-parallel: true - trust-remote-code: true - enable-prefix-caching: true - kv-cache-metrics: true - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - block-size: 128 - gpu-memory-utilization: 0.95 - max-model-len: 1048576 - language-model-only: true - kv-cache-dtype: "fp8" - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":549755813888,"cpu_bytes_to_use_per_rank":137438953472,"lazy_offload":false}}' - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY"}' - stream-interval: 20 - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - kv_events: true -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/srt_agentic.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-dep4.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-dep4.yaml deleted file mode 100644 index 8bb79b49b4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-dep4.yaml +++ /dev/null @@ -1,106 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-agg-gb200-dep4-agentic" - -model: {path: "minimax-m3-nvfp4", container: "vllm/vllm-openai:v0.27.1", precision: "fp4"} -identity: - model: {repo: "nvidia/MiniMax-M3-NVFP4"} - container: {image: "vllm/vllm-openai:v0.27.1"} - frameworks: {dynamo: "1.3.1"} -dynamo: {install: true, source: {pypi: "1.3.1"}} -environment: {ETCD_LEASE_TTL: "7200"} -slurm: {time_limit: "12:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} -resources: {gpu_type: "gb200", gpus_per_node: 4} -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - env: {DYN_TCP_REQUEST_TIMEOUT: "60"} - args: - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "minimax_m3" - reasoning-parser: "minimax_m3" - enable-auto-tool-choice: true - router-mode: "kv" - router-kv-events: true - router-reset-states: true - router-temperature: "0" - router-session-affinity-ttl-secs: 14400 - kv-cache-block-size: 128 -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - env: - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_LOG_STATS_INTERVAL: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "cuda_copy,cuda_ipc,rc" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 4 - data-parallel-rpc-port: 13345 - enable-expert-parallel: true - trust-remote-code: true - enable-prefix-caching: true - kv-cache-metrics: true - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - block-size: 128 - gpu-memory-utilization: 0.95 - max-model-len: 1048576 - language-model-only: true - kv-cache-dtype: "fp8" - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY"}' - stream-interval: 20 - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - kv_events: true -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/srt_agentic.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-dep8.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-dep8.yaml deleted file mode 100644 index 49fda02e0c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-dep8.yaml +++ /dev/null @@ -1,106 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-agg-gb200-dep8-agentic" - -model: {path: "minimax-m3-nvfp4", container: "vllm/vllm-openai:v0.27.1", precision: "fp4"} -identity: - model: {repo: "nvidia/MiniMax-M3-NVFP4"} - container: {image: "vllm/vllm-openai:v0.27.1"} - frameworks: {dynamo: "1.3.1"} -dynamo: {install: true, source: {pypi: "1.3.1"}} -environment: {ETCD_LEASE_TTL: "7200"} -slurm: {time_limit: "12:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} -resources: {gpu_type: "gb200", gpus_per_node: 4} -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - env: {DYN_TCP_REQUEST_TIMEOUT: "60"} - args: - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "minimax_m3" - reasoning-parser: "minimax_m3" - enable-auto-tool-choice: true - router-mode: "kv" - router-kv-events: true - router-reset-states: true - router-temperature: "0" - router-session-affinity-ttl-secs: 14400 - kv-cache-block-size: 128 -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - env: - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_LOG_STATS_INTERVAL: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "cuda_copy,cuda_ipc,rc" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - data-parallel-rpc-port: 13345 - enable-expert-parallel: true - trust-remote-code: true - enable-prefix-caching: true - kv-cache-metrics: true - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - block-size: 128 - gpu-memory-utilization: 0.95 - max-model-len: 1048576 - language-model-only: true - kv-cache-dtype: "fp8" - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY"}' - stream-interval: 20 - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - kv_events: true -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/srt_agentic.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-vllm-simple.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-vllm-simple.yaml deleted file mode 100644 index 54390af54e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-vllm-simple.yaml +++ /dev/null @@ -1,113 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-agg-gb200-tp4-vllm-simple-agentic" - -model: - path: "minimax-m3-nvfp4" - container: "vllm/vllm-openai:v0.27.1" - precision: "fp4" - -identity: - model: {repo: "nvidia/MiniMax-M3-NVFP4"} - container: {image: "vllm/vllm-openai:v0.27.1"} - frameworks: {dynamo: "1.3.1"} - -dynamo: {install: true, source: {pypi: "1.3.1"}} -environment: {ETCD_LEASE_TTL: "7200"} -slurm: {time_limit: "12:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} -resources: {gpu_type: "gb200", gpus_per_node: 4} -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - -frontend: - type: dynamo - enable_multiple_frontends: false - env: {DYN_TCP_REQUEST_TIMEOUT: "60"} - args: - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "minimax_m3" - reasoning-parser: "minimax_m3" - enable-auto-tool-choice: true - router-mode: "kv" - router-kv-events: true - router-reset-states: true - router-temperature: "0" - router-session-affinity-ttl-secs: 14400 - kv-cache-block-size: 128 - -engine: - type: vllm - connector: -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - env: - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_USE_SIMPLE_KV_OFFLOAD: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "cuda_copy,cuda_ipc,rc" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 4 - pipeline-parallel-size: 1 - trust-remote-code: true - enable-prefix-caching: true - kv-cache-metrics: true - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - block-size: 128 - gpu-memory-utilization: 0.9 - max-model-len: 1048576 - language-model-only: true - kv-cache-dtype: "fp8" - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":549755813888,"cpu_bytes_to_use_per_rank":137438953472,"lazy_offload":false}}' - stream-interval: 20 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY"}' - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - - kv_events: true -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/srt_agentic.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4.yaml deleted file mode 100644 index e956a39828..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4.yaml +++ /dev/null @@ -1,116 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-agg-gb200-tp4-agentic" - -model: - path: "minimax-m3-nvfp4" - container: "vllm/vllm-openai:v0.27.1" - precision: "fp4" - -identity: - model: {repo: "nvidia/MiniMax-M3-NVFP4"} - container: {image: "vllm/vllm-openai:v0.27.1"} - frameworks: {dynamo: "1.3.1"} - -dynamo: {install: true, source: {pypi: "1.3.1"}} -environment: {ETCD_LEASE_TTL: "7200"} - -slurm: {time_limit: "12:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - -frontend: - type: dynamo - enable_multiple_frontends: false - env: {DYN_TCP_REQUEST_TIMEOUT: "60"} - args: - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "minimax_m3" - reasoning-parser: "minimax_m3" - enable-auto-tool-choice: true - router-mode: "kv" - router-kv-events: true - router-reset-states: true - router-temperature: "0" - router-session-affinity-ttl-secs: 14400 - kv-cache-block-size: 128 - -engine: - type: vllm - connector: -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - - env: - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_LOG_STATS_INTERVAL: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "cuda_copy,cuda_ipc,rc" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 4 - pipeline-parallel-size: 1 - trust-remote-code: true - enable-prefix-caching: true - kv-cache-metrics: true - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY"}' - block-size: 128 - gpu-memory-utilization: 0.9 - max-model-len: 1048576 - language-model-only: true - kv-cache-dtype: "fp8" - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - stream-interval: 20 - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - - kv_events: true -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/srt_agentic.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep4.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep4.yaml deleted file mode 100644 index d7d6023eaf..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep4.yaml +++ /dev/null @@ -1,142 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb200-1p1d-dep8-dep4-agentic" - -model: {path: "minimax-m3-nvfp4", container: "vllm/vllm-openai:v0.27.1", precision: "fp4"} -identity: - model: {repo: "nvidia/MiniMax-M3-NVFP4"} - container: {image: "vllm/vllm-openai:v0.27.1"} - frameworks: {dynamo: "1.3.1"} -dynamo: {install: true, source: {pypi: "1.3.1"}} -environment: {ETCD_LEASE_TTL: "7200"} -slurm: {time_limit: "12:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - env: {DYN_TCP_REQUEST_TIMEOUT: "60"} - args: - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "minimax_m3" - reasoning-parser: "minimax_m3" - enable-auto-tool-choice: true - router-mode: "kv" - router-kv-events: true - router-reset-states: true - router-temperature: "0" - router-session-affinity-ttl-secs: 14400 - kv-cache-block-size: 128 -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - env: &worker_env - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_LOG_STATS_INTERVAL: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "cuda_copy,cuda_ipc,rc" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - data-parallel-rpc-port: 13345 - enable-expert-parallel: true - trust-remote-code: true - enable-prefix-caching: true - kv-cache-metrics: true - kv-transfer-config: '{"kv_connector":"NixlConnector","kv_role":"kv_both"}' - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - block-size: 128 - gpu-memory-utilization: 0.95 - max-model-len: 1048576 - language-model-only: true - kv-cache-dtype: "fp8" - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - compilation-config: '{"cudagraph_mode":"PIECEWISE"}' - stream-interval: 20 - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - kv_events: true - decode: - nodes: 1 - workers: 1 - gpus: 4 - env: *worker_env - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 4 - data-parallel-rpc-port: 13345 - enable-expert-parallel: true - trust-remote-code: true - enable-prefix-caching: true - kv-cache-metrics: true - kv-transfer-config: '{"kv_connector":"NixlConnector","kv_role":"kv_both"}' - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - block-size: 128 - gpu-memory-utilization: 0.95 - max-model-len: 1048576 - language-model-only: true - kv-cache-dtype: "fp8" - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY"}' - stream-interval: 20 - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - kv_events: true -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/srt_agentic.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-hicache.yaml deleted file mode 100644 index c0077e00fb..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-hicache.yaml +++ /dev/null @@ -1,88 +0,0 @@ -schema: 2 -name: qwen35-gb200-sglang-agentic-mtp-agg-tp4-hicache - -model: {path: qwen3.5-fp4, container: dynamo-sglang, precision: fp4} -identity: - model: {repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2} - container: {image: lmsysorg/sglang:v0.5.17-cu130} -slurm: {time_limit: "8:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} -resources: {gpu_type: gb200, gpus_per_node: 4} -frontend: - type: sglang-router - args: - worker-startup-timeout-secs: 3600 - -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - env: - PYTHONNOUSERSITE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - SGLANG_ENABLE_SPEC_V2: "1" - SGL_ENABLE_JIT_DEEPGEMM: "false" - TORCH_CUDA_ARCH_LIST: "10.0" - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-symm-mem: true - quantization: modelopt_fp4 - fp4-gemm-backend: flashinfer_cutlass - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 8192 - attention-backend: trtllm_mha - linear-attn-decode-backend: flashinfer - moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs: 64 - max-running-requests: 160 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.78 - max-mamba-cache-size: 360 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - tokenizer-worker-num: 6 - page-size: 64 - enable-hierarchical-cache: true - hicache-ratio: 0.70 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-write-policy: write_back - mamba-max-states-per-path: 1 - enable-metrics: true - enable-cache-report: true - -sbatch_directives: {mem: "0", cpus-per-task: "144"} -srun_options: {mem: "0", container-remap-root: ""} -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/srt_agentic.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "false" - TP: "4" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" diff --git a/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4-mtp/agentic.yaml b/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4-mtp/agentic.yaml deleted file mode 100644 index 0621caa8e2..0000000000 --- a/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4-mtp/agentic.yaml +++ /dev/null @@ -1,185 +0,0 @@ -# Kimi-K3 MXFP4 AgentX on MI355X with vLLM DSpark -# (https://recipes.vllm.ai/moonshotai/Kimi-K3). TP8 only: the 1.56 TB checkpoint -# is ~195 GB per GPU. The KV cache is GPU-resident through c4 and backed by -# vLLM's SimpleCPUOffloadConnector from c8. The DCP8 arm (c44-c70) runs without -# a draft model and stays on the legacy script. -base: - schema: 2 - name: kimik3-fp4-mi355x-vllm-agentic - model: - path: hf:moonshotai/Kimi-K3 - container: vllm/vllm-openai-rocm:nightly-rocm100-af1c01499b289be555c475669ba50a88e96d846e - precision: fp4 - resources: - gpu_type: mi355x - gpus_per_node: 8 - frontend: - type: vllm - enable_multiple_frontends: false - observability: - enabled: false - tachometer: - enabled: false - engine: - type: vllm - connector: null - # Weights load for up to VLLM_ENGINE_READY_TIMEOUT_S. - health_check: - interval_seconds: 10 - max_attempts: 720 - roles: - agg: - nodes: 1 - workers: 1 - gpus: 8 - args: - served-model-name: moonshotai/Kimi-K3 - trust-remote-code: true - moe-backend: auto - tensor-parallel-size: 8 - load-format: fastsafetensors - gpu-memory-utilization: 0.9 - language-model-only: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - max-model-len: 1048576 - enable-prefix-caching: true - kv-cache-dtype: fp8 - attention-config: '{"mla_prefill_backend":"ROCM_AITER_FA"}' - env: - # Upstream AMD recipe environment. - VLLM_ROCM_AITER_MLA_ASM_PADDING: asm - VLLM_ROCM_USE_AITER: '1' - SAFETENSORS_FAST_GPU: '1' - VLLM_ROCM_USE_AITER_MOE_SITUV2_A8W4: '1' - AITER_SITUV2_A8W4: '1' - AITER_BF16_FP8_MOE_BOUND: '0' - AITER_QUICK_REDUCE_QUANTIZATION: INT4 - # The MI355X nodes report MEC firmware 38, below the 177 that fixes the - # RCCL memory reclaim issue. - HSA_NO_SCRATCH_RECLAIM: '1' - # 2.8 TB of weights off a shared mount takes far longer than the default. - VLLM_ENGINE_READY_TIMEOUT_S: '7200' - PYTHONNOUSERSITE: '1' - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1200' - VLLM_USE_DIRECT_DCP_A2A: '0' - VLLM_USE_DIRECT_DCP_Q_GATHER: '0' - VLLM_USE_DIRECT_DCP_KV_GATHER: '0' - benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/srt_agentic.sh - env: - MODEL: moonshotai/Kimi-K3 - # Long agentic turns against a 1M context are prefill-bound on the server. - AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' - -# One variant per point. The DSpark draft length descends with concurrency -# (7, 5, 4, 4, 3, 3); throughput runs replace block rejection with the golden -# acceptance length for that draft length. Admission is 2x CONC and graph -# capture covers every size up to 2x CONC x (1 + drafts). Breakable -# FULL_AND_PIECEWISE graphs cost KV pool, so only c1 and c4 use them. The -# offload points split the 1799 GB DRAM budget evenly across the eight ranks; -# identical prefixes must hash to identical block keys on every rank. -override_tp8_c1: - roles: - agg: - args: - max-num-seqs: 2 - max-num-batched-tokens: 16384 - compilation-config: '{"mode":3,"cudagraph_mode":"FULL_AND_PIECEWISE","max_cudagraph_capture_size":16,"custom_ops":["+fused_rms_norm_gated"],"cudagraph_capture_sizes":[2,3,4,5,6,7,8,9,10,11,12,13,14,15,16]}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"method":"dspark","attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - env: - VLLM_USE_BREAKABLE_CUDAGRAPH: '1' - benchmark: - env: - CONC: '1' - KV_OFFLOADING: 'none' - -override_tp8_c4: - roles: - agg: - args: - max-num-seqs: 8 - max-num-batched-tokens: 8192 - compilation-config: '{"mode":3,"cudagraph_mode":"FULL_AND_PIECEWISE","max_cudagraph_capture_size":48,"custom_ops":["+fused_rms_norm_gated"],"cudagraph_capture_sizes":[2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48]}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":5,"method":"dspark","attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - env: - VLLM_USE_BREAKABLE_CUDAGRAPH: '1' - benchmark: - env: - CONC: '4' - KV_OFFLOADING: 'none' - -override_tp8_c8_simple: - roles: - agg: - args: - max-num-seqs: 16 - max-num-batched-tokens: 8192 - compilation-config: '{"mode":3,"cudagraph_mode":"FULL","max_cudagraph_capture_size":80,"custom_ops":["+fused_rms_norm_gated"],"cudagraph_capture_sizes":[2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80]}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":4,"method":"dspark","attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use_per_rank":224875000000,"lazy_offload":false}}' - env: - VLLM_USE_BREAKABLE_CUDAGRAPH: '0' - PYTHONHASHSEED: '42' - benchmark: - env: - CONC: '8' - KV_OFFLOADING: 'dram' - TOTAL_CPU_DRAM_GB: '1799' - -override_tp8_c10_simple: - roles: - agg: - args: - max-num-seqs: 20 - max-num-batched-tokens: 8192 - compilation-config: '{"mode":3,"cudagraph_mode":"FULL","max_cudagraph_capture_size":100,"custom_ops":["+fused_rms_norm_gated"],"cudagraph_capture_sizes":[2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100]}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":4,"method":"dspark","attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use_per_rank":224875000000,"lazy_offload":false}}' - env: - VLLM_USE_BREAKABLE_CUDAGRAPH: '0' - PYTHONHASHSEED: '42' - benchmark: - env: - CONC: '10' - KV_OFFLOADING: 'dram' - TOTAL_CPU_DRAM_GB: '1799' - -override_tp8_c12_simple: - roles: - agg: - args: - max-num-seqs: 24 - max-num-batched-tokens: 8192 - compilation-config: '{"mode":3,"cudagraph_mode":"FULL","max_cudagraph_capture_size":96,"custom_ops":["+fused_rms_norm_gated"],"cudagraph_capture_sizes":[2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96]}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":3,"method":"dspark","attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use_per_rank":224875000000,"lazy_offload":false}}' - env: - VLLM_USE_BREAKABLE_CUDAGRAPH: '0' - PYTHONHASHSEED: '42' - benchmark: - env: - CONC: '12' - KV_OFFLOADING: 'dram' - TOTAL_CPU_DRAM_GB: '1799' - -override_tp8_c14_simple: - roles: - agg: - args: - max-num-seqs: 28 - max-num-batched-tokens: 8192 - compilation-config: '{"mode":3,"cudagraph_mode":"FULL","max_cudagraph_capture_size":112,"custom_ops":["+fused_rms_norm_gated"],"cudagraph_capture_sizes":[2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112]}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":3,"method":"dspark","attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use_per_rank":224875000000,"lazy_offload":false}}' - env: - VLLM_USE_BREAKABLE_CUDAGRAPH: '0' - PYTHONHASHSEED: '42' - benchmark: - env: - CONC: '14' - KV_OFFLOADING: 'dram' - TOTAL_CPU_DRAM_GB: '1799' diff --git a/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi355x-fp4-mtp/agentic.yaml b/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi355x-fp4-mtp/agentic.yaml deleted file mode 100644 index d9c5e537a7..0000000000 --- a/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi355x-fp4-mtp/agentic.yaml +++ /dev/null @@ -1,204 +0,0 @@ -# MiniMax-M3 MXFP4 AgentX on MI355X with vLLM EAGLE3 (GQA draft). The KV cache -# is GPU-resident. -base: - schema: 2 - name: minimaxm3-fp4-mi355x-vllm-agentic - model: - path: hf:amd/MiniMax-M3-MXFP4 - container: vllm/vllm-openai-rocm:nightly-2a02f6efe319c885e3ccbcecde402e0028f9ec1e - precision: fp4 - resources: - gpu_type: mi355x - gpus_per_node: 8 - frontend: - type: vllm - enable_multiple_frontends: false - observability: - enabled: false - tachometer: - enabled: false - engine: - type: vllm - connector: null - # The engine may take the full VLLM_ENGINE_READY_TIMEOUT_S to become ready. - health_check: - interval_seconds: 10 - max_attempts: 360 - roles: - agg: - nodes: 1 - workers: 1 - args: - served-model-name: amd/MiniMax-M3-MXFP4 - trust-remote-code: true - block-size: 128 - gpu-memory-utilization: 0.90 - enable-chunked-prefill: true - max-num-batched-tokens: 32768 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes": [1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,20,22,24,26,28,30,32,34,36,48,64,72,80,88,96,104,112,120,128]}' - language-model-only: true - enable-prefix-caching: true - attention-backend: ROCM_AITER_UNIFIED_ATTN - moe-backend: aiter - kv-cache-dtype: fp8 - attention-config: '{"indexer_kv_dtype": "fp8"}' - tool-call-parser: minimax_m3 - reasoning-parser: minimax_m3 - enable-auto-tool-choice: true - default-chat-template-kwargs: '{"thinking_mode":"enabled"}' - stream-interval: 20 - # Three-token EAGLE3; throughput runs replace verification with the - # golden acceptance length. - speculative-config: '{"method": "eagle3", "model": "Inferact/MiniMax-M3-EAGLE3-GQA", "num_speculative_tokens": 3, "attention_backend": "ROCM_AITER_UNIFIED_ATTN"}' - env: - PYTHONNOUSERSITE: '1' - VLLM_ENGINE_READY_TIMEOUT_S: '3600' - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' - VLLM_USE_BREAKABLE_CUDAGRAPH: '0' - VLLM_ROCM_USE_AITER: '1' - VLLM_ROCM_USE_AITER_MOE: '1' - VLLM_ROCM_USE_AITER_FUSION_SHARED_EXPERTS: '1' - VLLM_ROCM_SHUFFLE_KV_CACHE_LAYOUT: '1' - VLLM_ROCM_QUICK_REDUCE_QUANTIZATION: INT4 - VLLM_ROCM_QUICK_REDUCE_CAST_BF16_TO_FP16: '0' - VLLM_ROCM_QUICK_REDUCE_QUANTIZATION_MIN_SIZE_KB: '256' - benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/srt_agentic.sh - env: - MODEL: amd/MiniMax-M3-MXFP4 - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' - AIPERF_APPLY_CHAT_TEMPLATE: 'true' - -# One variant per point. Admission is 2x CONC. -override_tp4_c1: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-num-seqs: 2 - benchmark: - env: - CONC: '1' - -override_tp4_c4: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-num-seqs: 8 - benchmark: - env: - CONC: '4' - -override_tp4_c5: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-num-seqs: 10 - benchmark: - env: - CONC: '5' - -override_tp4_c8: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-num-seqs: 16 - benchmark: - env: - CONC: '8' - -override_tp4_c10: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-num-seqs: 20 - benchmark: - env: - CONC: '10' - -override_tp4_c12: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-num-seqs: 24 - benchmark: - env: - CONC: '12' - -override_tp4_c15: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-num-seqs: 30 - benchmark: - env: - CONC: '15' - -override_tp4_c20: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-num-seqs: 40 - benchmark: - env: - CONC: '20' - -override_tp4_c24: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-num-seqs: 48 - benchmark: - env: - CONC: '24' - -override_tp4_c32: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-num-seqs: 64 - benchmark: - env: - CONC: '32' - -override_tp2_c1: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - max-num-seqs: 2 - benchmark: - env: - CONC: '1' - -override_tp2_c2: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - max-num-seqs: 4 - benchmark: - env: - CONC: '2'