diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p1d-dep4-dep8-c308-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p1d-dep4-dep8-c308-stp.yaml deleted file mode 100644 index 7b70090991..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p1d-dep4-dep8-c308-stp.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: disagg-B200-1p1d-dep4-dep8-c308-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 1 - workers: 1 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.92 - max_batch_size: 32 - max_num_tokens: 32 - max_seq_len: 9256 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 308 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c24-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c24-stp.yaml deleted file mode 100644 index 9efed38c1d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c24-stp.yaml +++ /dev/null @@ -1,127 +0,0 @@ -schema: 2 -name: disagg-B200-1p4d-dep4-tep8-c24-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 4 - workers: 4 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 4 - max_num_tokens: 4 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 24 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c4-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c4-stp.yaml deleted file mode 100644 index 33e18fd915..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p4d-dep4-tep8-c4-stp.yaml +++ /dev/null @@ -1,127 +0,0 @@ -schema: 2 -name: disagg-B200-1p4d-dep4-tep8-c4-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 4 - workers: 4 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 1 - max_num_tokens: 1 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 4 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c115-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c115-stp.yaml deleted file mode 100644 index b03bbea144..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c115-stp.yaml +++ /dev/null @@ -1,129 +0,0 @@ -schema: 2 -name: disagg-B200-1p5d-dep4-tep4-c115-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 3 - workers: 5 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 16 - max_num_tokens: 16 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 4 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 115 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c195-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c195-stp.yaml deleted file mode 100644 index 82ddd6648c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c195-stp.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: disagg-B200-1p5d-dep4-tep4-c195-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 3 - workers: 5 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 32 - max_num_tokens: 32 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 4 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 195 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c30-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c30-stp.yaml deleted file mode 100644 index 939cefcefd..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c30-stp.yaml +++ /dev/null @@ -1,127 +0,0 @@ -schema: 2 -name: disagg-B200-1p5d-dep4-tep4-c30-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 3 - workers: 5 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 4 - max_num_tokens: 4 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 4 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 30 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c5-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c5-stp.yaml deleted file mode 100644 index 8638021360..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c5-stp.yaml +++ /dev/null @@ -1,127 +0,0 @@ -schema: 2 -name: disagg-B200-1p5d-dep4-tep4-c5-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 3 - workers: 5 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 1 - max_num_tokens: 1 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 4 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 5 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c60-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c60-stp.yaml deleted file mode 100644 index c5cc9ceab4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-1p5d-dep4-tep4-c60-stp.yaml +++ /dev/null @@ -1,128 +0,0 @@ -schema: 2 -name: disagg-B200-1p5d-dep4-tep4-c60-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 3 - workers: 5 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 8 - max_num_tokens: 8 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 4 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 60 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-2p1d-dep4-dep8-c615-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-2p1d-dep4-dep8-c615-stp.yaml deleted file mode 100644 index 028a790246..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-2p1d-dep4-dep8-c615-stp.yaml +++ /dev/null @@ -1,135 +0,0 @@ -schema: 2 -name: disagg-B200-2p1d-dep4-dep8-c615-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 2 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 1 - workers: 1 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.92 - max_batch_size: 64 - max_num_tokens: 64 - max_seq_len: 9256 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 615 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-3p1d-dep4-dep8-c1127-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-3p1d-dep4-dep8-c1127-stp.yaml deleted file mode 100644 index 503f6c83a8..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-3p1d-dep4-dep8-c1127-stp.yaml +++ /dev/null @@ -1,143 +0,0 @@ -schema: 2 -name: disagg-B200-3p1d-dep4-dep8-c1127-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 3 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 1 - workers: 1 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.92 - max_batch_size: 128 - max_num_tokens: 128 - max_seq_len: 9256 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 1127 - req_rate: inf - num_prompts_mult: 20 diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-4p1d-dep4-dep8-c2151-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-4p1d-dep4-dep8-c2151-stp.yaml deleted file mode 100644 index d7a6aad22c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik2.5/trtllm/b200-fp4/8k1k/disagg-4p1d-dep4-dep8-c2151-stp.yaml +++ /dev/null @@ -1,159 +0,0 @@ -schema: 2 -name: disagg-B200-4p1d-dep4-dep8-c2151-stp -model: - path: kimik2.5-fp4 - container: dynamo-trtllm - precision: fp4 -identity: - model: - repo: nvidia/Kimi-K2.5-NVFP4 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc21 -slurm: - time_limit: 04:00:00 -dynamo: - install: true - source: - rev: 4025cb6a8992d19fd19eead36065e08ea5301f35 - request_plane: tcp -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: b200 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 4 - gpus: 4 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 16386 - max_seq_len: 8232 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - trust_remote_code: true - decode: - nodes: 1 - workers: 1 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: rc,cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: DEFAULT - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 68736 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - - 136 - - 144 - - 152 - - 160 - - 168 - - 176 - - 184 - - 192 - - 200 - - 208 - - 216 - - 224 - - 232 - - 240 - - 248 - - 256 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.92 - max_batch_size: 256 - max_num_tokens: 256 - max_seq_len: 9256 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 2151 - req_rate: inf - num_prompts_mult: 20