From 8fa3b4a3c46a04bcd7ca926abc503b5efa276356 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Tue, 28 Jul 2026 10:27:04 -0500 Subject: [PATCH 1/4] refactor: standardize srt-slurm on v1.0.36 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Read a single version pin from configs/srt-slurm-version.txt in every multinode launcher. Overlay InferenceX-owned recipes onto the tagged upstream checkout and migrate branch-only fields to the v1.0.36 schema. 中文:所有多节点启动器统一读取 configs/srt-slurm-version.txt 中的版本固定值。在上游标签检出上覆盖 InferenceX 自有配方,并将分支专用字段迁移到 v1.0.36 架构。 --- ...10dep4_gen1dep16_batch64_eplb384_mtp3.yaml | 153 +++++++++++++ ...12dep4_gen1dep8_batch512_eplb384_mtp1.yaml | 209 ++++++++++++++++++ .../ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml | 138 ++++++++++++ .../ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml | 138 ++++++++++++ .../ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml | 136 ++++++++++++ .../ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml | 138 ++++++++++++ ...tx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml | 145 ++++++++++++ ...tx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml | 145 ++++++++++++ ...tx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml | 146 ++++++++++++ ...x6dep4_gen1dep16_batch32_eplb384_mtp3.yaml | 149 +++++++++++++ ...x7dep4_gen1dep8_batch128_eplb384_mtp3.yaml | 161 ++++++++++++++ ...x8dep4_gen1dep32_batch16_eplb384_mtp3.yaml | 147 ++++++++++++ ...x9dep4_gen1dep8_batch256_eplb384_mtp1.yaml | 177 +++++++++++++++ ...10dep4_gen1dep8_batch512_eplb384_mtp0.yaml | 203 +++++++++++++++++ ...tx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml | 139 ++++++++++++ .../ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml | 132 +++++++++++ .../ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml | 132 +++++++++++ .../ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml | 132 +++++++++++ .../ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml | 132 +++++++++++ .../ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml | 133 +++++++++++ ...tx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml | 140 ++++++++++++ ...x4dep4_gen1dep32_batch16_eplb384_mtp0.yaml | 141 ++++++++++++ ...x5dep4_gen1dep16_batch64_eplb384_mtp0.yaml | 147 ++++++++++++ ...x6dep4_gen1dep32_batch32_eplb384_mtp0.yaml | 143 ++++++++++++ ...x6dep4_gen1dep8_batch256_eplb384_mtp0.yaml | 171 ++++++++++++++ ...9dep4_gen1dep16_batch128_eplb384_mtp0.yaml | 155 +++++++++++++ .../ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml | 128 +++++++++++ .../ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml | 130 +++++++++++ .../ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml | 131 +++++++++++ .../ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml | 129 +++++++++++ .../ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml | 129 +++++++++++ .../ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml | 130 +++++++++++ .../ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml | 130 +++++++++++ ...ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml | 137 ++++++++++++ ...ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml | 131 +++++++++++ ...ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml | 133 +++++++++++ ...tx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml | 144 ++++++++++++ ...tx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml | 161 ++++++++++++++ .../ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml | 130 +++++++++++ .../ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml | 130 +++++++++++ .../ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml | 130 +++++++++++ .../ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml | 131 +++++++++++ .../ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml | 129 +++++++++++ .../ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml | 130 +++++++++++ .../ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml | 130 +++++++++++ ...ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml | 131 +++++++++++ ...ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml | 133 +++++++++++ ...tx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml | 145 ++++++++++++ ...ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml | 137 ++++++++++++ .../8k1k/disagg-gb300-1p6d-dep4-tp4.yaml | 2 +- .../8k1k/disagg-gb300-1p9d-tep4-tp4.yaml | 2 +- .../disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml | 2 +- .../disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml | 2 +- .../disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml | 2 +- .../8k1k/disagg-gb300-7p2d-dep4-dep16.yaml | 2 +- .../disagg-gb200-2p1d-dep8-dep8-agentic.yaml | 1 - .../disagg-gb200-3p2d-tep8-tp8-agentic.yaml | 1 - configs/srt-slurm-version.txt | 1 + runners/launch_b200-dgxc.sh | 46 +--- runners/launch_b300-nv.sh | 54 +---- runners/launch_gb200-nv.sh | 91 +------- runners/launch_gb300-nv.sh | 105 +-------- runners/launch_h100-dgxc-slurm.sh | 19 +- runners/launch_h200-dgxc-slurm.sh | 19 +- runners/srt_slurm.sh | 67 ++++++ 65 files changed, 7055 insertions(+), 312 deletions(-) create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx10dep4_gen1dep16_batch64_eplb384_mtp3.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep8_batch512_eplb384_mtp1.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb384_mtp3.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep8_batch128_eplb384_mtp3.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep32_batch16_eplb384_mtp3.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx9dep4_gen1dep8_batch256_eplb384_mtp1.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx10dep4_gen1dep8_batch512_eplb384_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb384_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb384_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb384_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep8_batch256_eplb384_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb384_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml create mode 100644 benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml create mode 100644 configs/srt-slurm-version.txt create mode 100644 runners/srt_slurm.sh diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx10dep4_gen1dep16_batch64_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx10dep4_gen1dep16_batch64_eplb384_mtp3.yaml new file mode 100644 index 0000000000..ec300feb01 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx10dep4_gen1dep16_batch64_eplb384_mtp3.yaml @@ -0,0 +1,153 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx10dep4_gen1dep16_batch64_eplb384_mtp3" + +# ctx: 10 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP16/EP16, enable_attention_dp=true, max_batch=64 +# concurrency: 1229 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 10 + prefill_workers: 10 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 4 + gpus_per_decode: 16 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + tokens_per_block: 128 + max_batch_size: 64 + max_num_tokens: 256 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep16_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 16 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + stream_interval: 100 + tensor_parallel_size: 16 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "1229" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep8_batch512_eplb384_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep8_batch512_eplb384_mtp1.yaml new file mode 100644 index 0000000000..9504350976 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep8_batch512_eplb384_mtp1.yaml @@ -0,0 +1,209 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx12dep4_gen1dep8_batch512_eplb384_mtp1" + +# ctx: 12 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP8/EP8, enable_attention_dp=true, max_batch=512 +# concurrency: 4301 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 12 + prefill_workers: 12 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 2 + gpus_per_decode: 8 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 1 + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + - 72 + - 80 + - 88 + - 96 + - 104 + - 112 + - 120 + - 128 + - 136 + - 144 + - 152 + - 160 + - 168 + - 176 + - 184 + - 192 + - 200 + - 208 + - 216 + - 224 + - 232 + - 240 + - 248 + - 256 + - 264 + - 272 + - 280 + - 288 + - 296 + - 304 + - 312 + - 320 + - 328 + - 336 + - 344 + - 352 + - 360 + - 368 + - 376 + - 384 + - 392 + - 400 + - 408 + - 416 + - 424 + - 432 + - 440 + - 448 + - 456 + - 464 + - 472 + - 480 + - 488 + - 496 + - 504 + - 512 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.8 + tokens_per_block: 128 + max_batch_size: 512 + max_num_tokens: 1024 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep8_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 8 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 1 + stream_interval: 100 + tensor_parallel_size: 8 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "4301" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml new file mode 100644 index 0000000000..967b62e73c --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml @@ -0,0 +1,138 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx1dep4_gen4tep8_batch1_eplb0_mtp3" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 4 decode workers, TP8/EP8, max_batch=1 +# concurrency: 8 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" +resources: + gpu_type: "gb300" + + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + decode_workers: 4 + decode_nodes: 8 + gpus_per_decode: 8 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + enable_padding: true + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + tokens_per_block: 128 + max_batch_size: 1 + max_num_tokens: 4 + max_seq_len: 9256 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + moe_expert_parallel_size: 8 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + stream_interval: 100 + tensor_parallel_size: 8 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "8" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml new file mode 100644 index 0000000000..d76b44a199 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml @@ -0,0 +1,138 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch1_eplb0_mtp3" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, max_batch=1 +# concurrency: 10 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" +resources: + gpu_type: "gb300" + + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + enable_padding: true + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + tokens_per_block: 128 + max_batch_size: 1 + max_num_tokens: 4 + max_seq_len: 9256 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + moe_expert_parallel_size: 4 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + stream_interval: 100 + tensor_parallel_size: 4 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "10" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml new file mode 100644 index 0000000000..7621c206cf --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml @@ -0,0 +1,136 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch2_eplb0_mtp3" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, max_batch=2 +# concurrency: 15 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +resources: + gpu_type: "gb300" + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + enable_padding: true + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8 + max_seq_len: 9256 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + moe_expert_parallel_size: 4 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + stream_interval: 100 + tensor_parallel_size: 4 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "15" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml new file mode 100644 index 0000000000..1f67c008d7 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml @@ -0,0 +1,138 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch4_eplb0_mtp3" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, max_batch=4 +# concurrency: 30 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" +resources: + gpu_type: "gb300" + + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + enable_padding: true + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + tokens_per_block: 128 + max_batch_size: 4 + max_num_tokens: 16 + max_seq_len: 9256 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + moe_expert_parallel_size: 4 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + stream_interval: 100 + tensor_parallel_size: 4 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "30" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml new file mode 100644 index 0000000000..edd3854350 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml @@ -0,0 +1,145 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx2dep4_gen1dep32_batch2_eplb384_mtp3" + +# ctx: 2 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true, max_batch=2 +# concurrency: 84 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 2 + prefill_workers: 2 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8 + max_seq_len: 9256 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep32_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 32 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + stream_interval: 100 + tensor_parallel_size: 32 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "84" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml new file mode 100644 index 0000000000..2bc2cbdd4f --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml @@ -0,0 +1,145 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx3dep4_gen1dep32_batch4_eplb384_mtp3" + +# ctx: 3 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true, max_batch=4 +# concurrency: 180 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 3 + prefill_workers: 3 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + tokens_per_block: 128 + max_batch_size: 4 + max_num_tokens: 16 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep32_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 32 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + stream_interval: 100 + tensor_parallel_size: 32 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "180" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml new file mode 100644 index 0000000000..3beac66efc --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml @@ -0,0 +1,146 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx4dep4_gen1dep32_batch8_eplb384_mtp3" + +# ctx: 4 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true, max_batch=8 +# concurrency: 333 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 4 + prefill_workers: 4 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + tokens_per_block: 128 + max_batch_size: 8 + max_num_tokens: 32 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep32_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 32 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + stream_interval: 100 + tensor_parallel_size: 32 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "333" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb384_mtp3.yaml new file mode 100644 index 0000000000..50cd26ac06 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb384_mtp3.yaml @@ -0,0 +1,149 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx6dep4_gen1dep16_batch32_eplb384_mtp3" + +# ctx: 6 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP16/EP16, enable_attention_dp=true, max_batch=32 +# concurrency: 666 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 6 + prefill_workers: 6 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 4 + gpus_per_decode: 16 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + tokens_per_block: 128 + max_batch_size: 32 + max_num_tokens: 128 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep16_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 16 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + stream_interval: 100 + tensor_parallel_size: 16 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "666" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep8_batch128_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep8_batch128_eplb384_mtp3.yaml new file mode 100644 index 0000000000..f03c2b0cf5 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep8_batch128_eplb384_mtp3.yaml @@ -0,0 +1,161 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx7dep4_gen1dep8_batch128_eplb384_mtp3" + +# ctx: 7 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP8/EP8, enable_attention_dp=true, max_batch=128 +# concurrency: 1229 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 7 + prefill_workers: 7 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 2 + gpus_per_decode: 8 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + - 72 + - 80 + - 88 + - 96 + - 104 + - 112 + - 120 + - 128 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.8 + tokens_per_block: 128 + max_batch_size: 128 + max_num_tokens: 512 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep8_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 8 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + stream_interval: 100 + tensor_parallel_size: 8 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "1229" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep32_batch16_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep32_batch16_eplb384_mtp3.yaml new file mode 100644 index 0000000000..5870e0e0ba --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep32_batch16_eplb384_mtp3.yaml @@ -0,0 +1,147 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx8dep4_gen1dep32_batch16_eplb384_mtp3" + +# ctx: 8 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true, max_batch=16 +# concurrency: 615 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 8 + prefill_workers: 8 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + tokens_per_block: 128 + max_batch_size: 16 + max_num_tokens: 64 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep32_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 32 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + stream_interval: 100 + tensor_parallel_size: 32 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "615" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx9dep4_gen1dep8_batch256_eplb384_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx9dep4_gen1dep8_batch256_eplb384_mtp1.yaml new file mode 100644 index 0000000000..ba65d9ee4e --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx9dep4_gen1dep8_batch256_eplb384_mtp1.yaml @@ -0,0 +1,177 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx9dep4_gen1dep8_batch256_eplb384_mtp1" + +# ctx: 9 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP8/EP8, enable_attention_dp=true, max_batch=256 +# concurrency: 2253 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 9 + prefill_workers: 9 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 2 + gpus_per_decode: 8 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 1 + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + - 72 + - 80 + - 88 + - 96 + - 104 + - 112 + - 120 + - 128 + - 136 + - 144 + - 152 + - 160 + - 168 + - 176 + - 184 + - 192 + - 200 + - 208 + - 216 + - 224 + - 232 + - 240 + - 248 + - 256 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.8 + tokens_per_block: 128 + max_batch_size: 256 + max_num_tokens: 512 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep8_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 8 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + speculative_config: + decoding_type: MTP + max_draft_len: 1 + stream_interval: 100 + tensor_parallel_size: 8 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "2253" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx10dep4_gen1dep8_batch512_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx10dep4_gen1dep8_batch512_eplb384_mtp0.yaml new file mode 100644 index 0000000000..0db4c6c834 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx10dep4_gen1dep8_batch512_eplb384_mtp0.yaml @@ -0,0 +1,203 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx10dep4_gen1dep8_batch512_eplb384_mtp0" + +# ctx: 10 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP8/EP8, enable_attention_dp=true, max_batch=512 +# concurrency: 4301 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 10 + prefill_workers: 10 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 2 + gpus_per_decode: 8 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + - 72 + - 80 + - 88 + - 96 + - 104 + - 112 + - 120 + - 128 + - 136 + - 144 + - 152 + - 160 + - 168 + - 176 + - 184 + - 192 + - 200 + - 208 + - 216 + - 224 + - 232 + - 240 + - 248 + - 256 + - 264 + - 272 + - 280 + - 288 + - 296 + - 304 + - 312 + - 320 + - 328 + - 336 + - 344 + - 352 + - 360 + - 368 + - 376 + - 384 + - 392 + - 400 + - 408 + - 416 + - 424 + - 432 + - 440 + - 448 + - 456 + - 464 + - 472 + - 480 + - 488 + - 496 + - 504 + - 512 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.8 + tokens_per_block: 128 + max_batch_size: 512 + max_num_tokens: 512 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep8_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 8 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + stream_interval: 100 + tensor_parallel_size: 8 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "4301" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml new file mode 100644 index 0000000000..edc9758f82 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml @@ -0,0 +1,139 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx1dep4_gen1dep32_batch4_eplb384_mtp0" + +# ctx: 1 prefill worker, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true, max_batch=4 +# concurrency: 154 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + tokens_per_block: 128 + max_batch_size: 4 + max_num_tokens: 4 + max_seq_len: 9256 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep32_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 32 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + stream_interval: 100 + tensor_parallel_size: 32 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "154" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml new file mode 100644 index 0000000000..a9f8989178 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml @@ -0,0 +1,132 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx1dep4_gen4tep8_batch1_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 4 decode workers, TP8/EP8, max_batch=1 +# concurrency: 4 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" +resources: + gpu_type: "gb300" + + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + decode_workers: 4 + decode_nodes: 8 + gpus_per_decode: 8 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + enable_padding: true + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + tokens_per_block: 128 + max_batch_size: 1 + max_num_tokens: 1 + max_seq_len: 9256 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + moe_expert_parallel_size: 8 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + stream_interval: 100 + tensor_parallel_size: 8 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "4" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml new file mode 100644 index 0000000000..1dbb0437fc --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml @@ -0,0 +1,132 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch1_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, max_batch=1 +# concurrency: 5 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" +resources: + gpu_type: "gb300" + + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + enable_padding: true + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + tokens_per_block: 128 + max_batch_size: 1 + max_num_tokens: 1 + max_seq_len: 9256 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + moe_expert_parallel_size: 4 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + stream_interval: 100 + tensor_parallel_size: 4 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "5" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml new file mode 100644 index 0000000000..1dc653de37 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml @@ -0,0 +1,132 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch2_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, max_batch=2 +# concurrency: 15 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" +resources: + gpu_type: "gb300" + + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + enable_padding: true + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 2 + max_seq_len: 9256 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + moe_expert_parallel_size: 4 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + stream_interval: 100 + tensor_parallel_size: 4 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "15" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml new file mode 100644 index 0000000000..d0131a0224 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml @@ -0,0 +1,132 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch4_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, max_batch=4 +# concurrency: 25 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" +resources: + gpu_type: "gb300" + + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + enable_padding: true + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + tokens_per_block: 128 + max_batch_size: 4 + max_num_tokens: 4 + max_seq_len: 9256 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + moe_expert_parallel_size: 4 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + stream_interval: 100 + tensor_parallel_size: 4 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "25" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml new file mode 100644 index 0000000000..7eeb25c066 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml @@ -0,0 +1,133 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch8_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, max_batch=8 +# concurrency: 55 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" +resources: + gpu_type: "gb300" + + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + enable_padding: true + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + tokens_per_block: 128 + max_batch_size: 8 + max_num_tokens: 8 + max_seq_len: 9256 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + moe_expert_parallel_size: 4 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + stream_interval: 100 + tensor_parallel_size: 4 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "55" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml new file mode 100644 index 0000000000..993c48615c --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml @@ -0,0 +1,140 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx2dep4_gen1dep32_batch8_eplb384_mtp0" + +# ctx: 2 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true, max_batch=8 +# concurrency: 308 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 2 + prefill_workers: 2 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + tokens_per_block: 128 + max_batch_size: 8 + max_num_tokens: 8 + max_seq_len: 9256 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep32_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 32 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + stream_interval: 100 + tensor_parallel_size: 32 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "308" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb384_mtp0.yaml new file mode 100644 index 0000000000..6bc263d352 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb384_mtp0.yaml @@ -0,0 +1,141 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx4dep4_gen1dep32_batch16_eplb384_mtp0" + +# ctx: 4 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true, max_batch=16 +# concurrency: 615 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 4 + prefill_workers: 4 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + tokens_per_block: 128 + max_batch_size: 16 + max_num_tokens: 16 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep32_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 32 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + stream_interval: 100 + tensor_parallel_size: 32 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "615" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb384_mtp0.yaml new file mode 100644 index 0000000000..9303cd94a4 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb384_mtp0.yaml @@ -0,0 +1,147 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx5dep4_gen1dep16_batch64_eplb384_mtp0" + +# ctx: 5 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP16/EP16, enable_attention_dp=true, max_batch=64 +# concurrency: 1229 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 5 + prefill_workers: 5 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 4 + gpus_per_decode: 16 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + tokens_per_block: 128 + max_batch_size: 64 + max_num_tokens: 64 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep16_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 16 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + stream_interval: 100 + tensor_parallel_size: 16 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "1229" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb384_mtp0.yaml new file mode 100644 index 0000000000..a7c79ce5f4 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb384_mtp0.yaml @@ -0,0 +1,143 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx6dep4_gen1dep32_batch32_eplb384_mtp0" + +# ctx: 6 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true, max_batch=32 +# concurrency: 1127 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 6 + prefill_workers: 6 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + tokens_per_block: 128 + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep32_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 32 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + stream_interval: 100 + tensor_parallel_size: 32 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "1127" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep8_batch256_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep8_batch256_eplb384_mtp0.yaml new file mode 100644 index 0000000000..6f545411a0 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep8_batch256_eplb384_mtp0.yaml @@ -0,0 +1,171 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx6dep4_gen1dep8_batch256_eplb384_mtp0" + +# ctx: 6 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP8/EP8, enable_attention_dp=true, max_batch=256 +# concurrency: 2253 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 6 + prefill_workers: 6 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 2 + gpus_per_decode: 8 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + - 72 + - 80 + - 88 + - 96 + - 104 + - 112 + - 120 + - 128 + - 136 + - 144 + - 152 + - 160 + - 168 + - 176 + - 184 + - 192 + - 200 + - 208 + - 216 + - 224 + - 232 + - 240 + - 248 + - 256 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.8 + tokens_per_block: 128 + max_batch_size: 256 + max_num_tokens: 256 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep8_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 8 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + stream_interval: 100 + tensor_parallel_size: 8 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "2253" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb384_mtp0.yaml new file mode 100644 index 0000000000..7d355c69df --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb384_mtp0.yaml @@ -0,0 +1,155 @@ +name: "dsv4pro_mxfp4_ISL8K_OSL1K_ctx9dep4_gen1dep16_batch128_eplb384_mtp0" + +# ctx: 9 prefill workers, TP4/EP4, EPLB: num_slots=384 +# gen: 1 decode worker, TP16/EP16, enable_attention_dp=true, max_batch=128 +# concurrency: 2253 + +model: + path: "deepseek-ai/DeepSeek-V4-Pro" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-deepseek-v4-dev.1" + precision: "fp4" + +extra_mount: + - "tmpfs:/dev/shm:size=100%" + - "configs/dsv4-moe-load-balancer-configs:/dsv4-eplb-configs" +resources: + gpu_type: "gb300" + + prefill_nodes: 9 + prefill_workers: 9 + gpus_per_prefill: 4 + + decode_workers: 1 + decode_nodes: 4 + gpus_per_decode: 16 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: "1" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "-1" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + decode_environment: + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + PYTHONUNBUFFERED: "1" + ENROOT_ALLOW_DEV: "yes" + MIMALLOC_PURGE_DELAY: "0" + NCCL_GRAPH_MIXING_SUPPORT: "0" + PYTHONWARNINGS: "ignore::DeprecationWarning:cutlass.cute.core" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + + trtllm_config: + prefill: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.5 + tokens_per_block: 128 + max_batch_size: 2 + max_num_tokens: 8192 + max_seq_len: 8232 + moe_config: + backend: TRTLLM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_ctx_ep4_384.yaml" + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + tensor_parallel_size: 4 + + decode: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 8192 + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + - 72 + - 80 + - 88 + - 96 + - 104 + - 112 + - 120 + - 128 + enable_padding: true + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.7 + tokens_per_block: 128 + max_batch_size: 128 + max_num_tokens: 128 + max_seq_len: 9256 + moe_config: + backend: MEGAMOE_DEEPGEMM + load_balancer: "/dsv4-eplb-configs/moe_load_balancer_gen_ep16_slots384.yaml" + use_low_precision_moe_combine: true + moe_expert_parallel_size: 16 + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + stream_interval: 100 + tensor_parallel_size: 16 + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "2253" + req_rate: "inf" + num_prompts_mult: 10 + num_warmup_mult: 20 + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" + use_chat_template: true + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 720 + interval_seconds: 10 + +dynamo: + request_plane: "tcp" + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml new file mode 100644 index 0000000000..55acb298f7 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml @@ -0,0 +1,128 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx1dep4_gen4tep8_batch1_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 4 decode workers, TP8/EP8, allreduce_strategy=MNNVL, enable_attention_dp=false +# STP (no speculative decoding) +# concurrency: 8 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb200" + + # Prefill: 1 worker x TP4 = 4 GPUs = 1 node + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + # Decode: 4 workers x TP8 = 32 GPUs = 8 nodes + decode_workers: 4 + decode_nodes: 8 + gpus_per_decode: 8 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + pipeline_parallel_size: 1 + allreduce_strategy: MNNVL + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 1 + max_num_tokens: 1 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "8" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml new file mode 100644 index 0000000000..3dcb33e574 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml @@ -0,0 +1,130 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx1dep4_gen4tep8_batch4_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 4 decode workers, TP8/EP8, allreduce_strategy=MNNVL, enable_attention_dp=false +# STP (no speculative decoding) +# concurrency: 24 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb200" + + # Prefill: 1 worker x TP4 = 4 GPUs = 1 node + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + # Decode: 4 workers x TP8 = 32 GPUs = 8 nodes + decode_workers: 4 + decode_nodes: 8 + gpus_per_decode: 8 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + pipeline_parallel_size: 1 + allreduce_strategy: MNNVL + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 4 + max_num_tokens: 4 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "24" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml new file mode 100644 index 0000000000..14aea33a88 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml @@ -0,0 +1,131 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch16_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, allreduce_strategy=MNNVL, enable_attention_dp=false +# STP (no speculative decoding) +# concurrency: 105 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb200" + + # Prefill: 1 worker x TP4 = 4 GPUs = 1 node + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + # Decode: 5 workers x TP4 = 20 GPUs = 5 nodes + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 16 + max_num_tokens: 16 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "105" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml new file mode 100644 index 0000000000..1d9b64c632 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml @@ -0,0 +1,129 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch1_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, allreduce_strategy=MNNVL, enable_attention_dp=false +# STP (no speculative decoding) +# concurrency: 5 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb200" + + # Prefill: 1 worker x TP4 = 4 GPUs = 1 node + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + # Decode: 5 workers x TP4 = 20 GPUs = 5 nodes + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 1 + max_num_tokens: 1 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "5" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml new file mode 100644 index 0000000000..eda6e401a1 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml @@ -0,0 +1,129 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch4_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, allreduce_strategy=MNNVL, enable_attention_dp=false +# STP (no speculative decoding) +# concurrency: 30 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb200" + + # Prefill: 1 worker x TP4 = 4 GPUs = 1 node + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + # Decode: 5 workers x TP4 = 20 GPUs = 5 nodes + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 4 + max_num_tokens: 4 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "30" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml new file mode 100644 index 0000000000..26d2973090 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml @@ -0,0 +1,130 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch8_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, allreduce_strategy=MNNVL, enable_attention_dp=false +# STP (no speculative decoding) +# concurrency: 60 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb200" + + # Prefill: 1 worker x TP4 = 4 GPUs = 1 node + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + # Decode: 5 workers x TP4 = 20 GPUs = 5 nodes + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 8 + max_num_tokens: 8 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "60" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml new file mode 100644 index 0000000000..10e9102bf2 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml @@ -0,0 +1,130 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx2dep4_gen1dep32_batch8_eplb0_mtp0" + +# ctx: 2 prefill workers, TP4/EP4 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true +# STP (no speculative decoding) +# concurrency: 333 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb200" + + # Prefill: 2 workers x TP4 = 8 GPUs = 2 nodes + prefill_nodes: 2 + prefill_workers: 2 + gpus_per_prefill: 4 + + # Decode: 1 worker x TP32 = 32 GPUs = 8 nodes + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + pipeline_parallel_size: 1 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 8 + max_num_tokens: 8 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "333" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml new file mode 100644 index 0000000000..34742abb62 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml @@ -0,0 +1,137 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx4dep4_gen1dep16_batch64_eplb0_mtp0" + +# ctx: 4 prefill workers, TP4/EP4 +# gen: 1 decode worker, TP16/EP16, enable_attention_dp=true +# STP (no speculative decoding) +# concurrency: 1229 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb200" + + # Prefill: 4 workers x TP4 = 16 GPUs = 4 nodes + prefill_nodes: 4 + prefill_workers: 4 + gpus_per_prefill: 4 + + # Decode: 1 worker x TP16 = 16 GPUs = 4 nodes + decode_workers: 1 + decode_nodes: 4 + gpus_per_decode: 16 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + pipeline_parallel_size: 1 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 64 + max_num_tokens: 64 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.8 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "1229" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml new file mode 100644 index 0000000000..4b47b44101 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml @@ -0,0 +1,131 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx4dep4_gen1dep32_batch16_eplb0_mtp0" + +# ctx: 4 prefill workers, TP4/EP4 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true +# STP (no speculative decoding) +# concurrency: 666 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb200" + + # Prefill: 4 workers x TP4 = 16 GPUs = 4 nodes + prefill_nodes: 4 + prefill_workers: 4 + gpus_per_prefill: 4 + + # Decode: 1 worker x TP32 = 32 GPUs = 8 nodes + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + pipeline_parallel_size: 1 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 16 + max_num_tokens: 16 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "666" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml new file mode 100644 index 0000000000..35d8337888 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml @@ -0,0 +1,133 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx6dep4_gen1dep32_batch32_eplb0_mtp0" + +# ctx: 6 prefill workers, TP4/EP4 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true +# STP (no speculative decoding) +# concurrency: 1229 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb200" + + # Prefill: 6 workers x TP4 = 24 GPUs = 6 nodes + prefill_nodes: 6 + prefill_workers: 6 + gpus_per_prefill: 4 + + # Decode: 1 worker x TP32 = 32 GPUs = 8 nodes + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + pipeline_parallel_size: 1 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "1229" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml new file mode 100644 index 0000000000..871b7429ba --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml @@ -0,0 +1,144 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx7dep4_gen1dep16_batch128_eplb0_mtp0" + +# ctx: 7 prefill workers, TP4/EP4 +# gen: 1 decode worker, TP16/EP16, enable_attention_dp=true, max_batch=128 +# concurrency: 2253 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb200" + + # Prefill: 7 workers x TP4 = 28 GPUs = 7 nodes + prefill_nodes: 7 + prefill_workers: 7 + gpus_per_prefill: 4 + + # Decode: 1 worker x TP16 = 16 GPUs = 4 nodes + decode_workers: 1 + decode_nodes: 4 + gpus_per_decode: 16 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + pipeline_parallel_size: 1 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 128 + max_num_tokens: 128 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + - 72 + - 80 + - 88 + - 96 + - 104 + - 112 + - 120 + - 128 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.8 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "2253" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml new file mode 100644 index 0000000000..fdd8c695dc --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml @@ -0,0 +1,161 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx9dep4_gen1dep16_batch256_eplb0_mtp0" + +# ctx: 9 prefill workers, TP4/EP4 +# gen: 1 decode worker, TP16/EP16, enable_attention_dp=true +# STP (no speculative decoding) +# concurrency: 4301 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb200" + + # Prefill: 9 workers x TP4 = 36 GPUs = 9 nodes + prefill_nodes: 9 + prefill_workers: 9 + gpus_per_prefill: 4 + + # Decode: 1 worker x TP16 = 16 GPUs = 4 nodes + decode_workers: 1 + decode_nodes: 4 + gpus_per_decode: 16 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + pipeline_parallel_size: 1 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 256 + max_num_tokens: 256 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + - 72 + - 80 + - 88 + - 96 + - 104 + - 112 + - 120 + - 128 + - 136 + - 144 + - 152 + - 160 + - 168 + - 176 + - 184 + - 192 + - 200 + - 208 + - 216 + - 224 + - 232 + - 240 + - 248 + - 256 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.8 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "4301" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml new file mode 100644 index 0000000000..9c56552fd1 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml @@ -0,0 +1,130 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx1dep4_gen4tep8_batch1_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 4 decode workers, TP8/EP8, allreduce_strategy=MNNVL, enable_attention_dp=false +# STP (no speculative decoding) +# concurrency: 4 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb300" + + # Prefill: 1 worker x TP4 = 4 GPUs = 1 node + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + # Decode: 4 workers x TP8 = 32 GPUs = 8 nodes + decode_workers: 4 + decode_nodes: 8 + gpus_per_decode: 8 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + pipeline_parallel_size: 1 + allreduce_strategy: MNNVL + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 1 + max_num_tokens: 1 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "4" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml new file mode 100644 index 0000000000..2df7b083ff --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml @@ -0,0 +1,130 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx1dep4_gen4tep8_batch2_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 4 decode workers, TP8/EP8, allreduce_strategy=MNNVL, enable_attention_dp=false +# STP (no speculative decoding) +# concurrency: 12 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb300" + + # Prefill: 1 worker x TP4 = 4 GPUs = 1 node + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + # Decode: 4 workers x TP8 = 32 GPUs = 8 nodes + decode_workers: 4 + decode_nodes: 8 + gpus_per_decode: 8 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + pipeline_parallel_size: 1 + allreduce_strategy: MNNVL + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 2 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "12" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml new file mode 100644 index 0000000000..04faad8e60 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml @@ -0,0 +1,130 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx1dep4_gen4tep8_batch4_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 4 decode workers, TP8/EP8, allreduce_strategy=MNNVL, enable_attention_dp=false +# STP (no speculative decoding) +# concurrency: 24 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb300" + + # Prefill: 1 worker x TP4 = 4 GPUs = 1 node + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + # Decode: 4 workers x TP8 = 32 GPUs = 8 nodes + decode_workers: 4 + decode_nodes: 8 + gpus_per_decode: 8 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + pipeline_parallel_size: 1 + allreduce_strategy: MNNVL + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 4 + max_num_tokens: 4 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "24" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml new file mode 100644 index 0000000000..e75f1fb65e --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml @@ -0,0 +1,131 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch16_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, allreduce_strategy=MNNVL, enable_attention_dp=false +# STP (no speculative decoding) +# concurrency: 115 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb300" + + # Prefill: 1 worker x TP4 = 4 GPUs = 1 node + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + # Decode: 5 workers x TP4 = 20 GPUs = 5 nodes + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 16 + max_num_tokens: 16 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "115" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml new file mode 100644 index 0000000000..0e5288717c --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml @@ -0,0 +1,129 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch4_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, allreduce_strategy=MNNVL, enable_attention_dp=false +# STP (no speculative decoding) +# concurrency: 30 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb300" + + # Prefill: 1 worker x TP4 = 4 GPUs = 1 node + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + # Decode: 5 workers x TP4 = 20 GPUs = 5 nodes + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 4 + max_num_tokens: 4 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "30" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml new file mode 100644 index 0000000000..0a58cf8279 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml @@ -0,0 +1,130 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx1dep4_gen5tep4_batch8_eplb0_mtp0" + +# ctx: 1 prefill worker, TP4/EP4 +# gen: 5 decode workers, TP4/EP4, allreduce_strategy=MNNVL, enable_attention_dp=false +# STP (no speculative decoding) +# concurrency: 60 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb300" + + # Prefill: 1 worker x TP4 = 4 GPUs = 1 node + prefill_nodes: 1 + prefill_workers: 1 + gpus_per_prefill: 4 + + # Decode: 5 workers x TP4 = 20 GPUs = 5 nodes + decode_workers: 5 + decode_nodes: 5 + gpus_per_decode: 4 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 8 + max_num_tokens: 8 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "60" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: false + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml new file mode 100644 index 0000000000..c2875c1215 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml @@ -0,0 +1,130 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx2dep4_gen1dep32_batch8_eplb0_mtp0" + +# ctx: 2 prefill workers, TP4/EP4 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true +# STP (no speculative decoding) +# concurrency: 333 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb300" + + # Prefill: 2 workers x TP4 = 8 GPUs = 2 nodes + prefill_nodes: 2 + prefill_workers: 2 + gpus_per_prefill: 4 + + # Decode: 1 worker x TP32 = 32 GPUs = 8 nodes + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + pipeline_parallel_size: 1 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 8 + max_num_tokens: 8 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "333" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml new file mode 100644 index 0000000000..ee066c1dfc --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml @@ -0,0 +1,131 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx3dep4_gen1dep32_batch16_eplb0_mtp0" + +# ctx: 3 prefill workers, TP4/EP4 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true +# STP (no speculative decoding) +# concurrency: 615 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb300" + + # Prefill: 3 workers x TP4 = 12 GPUs = 3 nodes + prefill_nodes: 3 + prefill_workers: 3 + gpus_per_prefill: 4 + + # Decode: 1 worker x TP32 = 32 GPUs = 8 nodes + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + pipeline_parallel_size: 1 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 16 + max_num_tokens: 16 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "615" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml new file mode 100644 index 0000000000..c28099c396 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml @@ -0,0 +1,133 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx5dep4_gen1dep32_batch32_eplb0_mtp0" + +# ctx: 5 prefill workers, TP4/EP4 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true +# STP (no speculative decoding) +# concurrency: 1229 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb300" + + # Prefill: 5 workers x TP4 = 20 GPUs = 5 nodes + prefill_nodes: 5 + prefill_workers: 5 + gpus_per_prefill: 4 + + # Decode: 1 worker x TP32 = 32 GPUs = 8 nodes + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + pipeline_parallel_size: 1 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 32 + max_num_tokens: 32 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "1229" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml new file mode 100644 index 0000000000..4da978b230 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml @@ -0,0 +1,145 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx6dep4_gen1dep16_batch128_eplb0_mtp0" + +# ctx: 6 prefill workers, TP4/EP4 +# gen: 1 decode worker, TP16/EP16, enable_attention_dp=true +# STP (no speculative decoding) +# concurrency: 2253 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb300" + + # Prefill: 6 workers x TP4 = 24 GPUs = 6 nodes + prefill_nodes: 6 + prefill_workers: 6 + gpus_per_prefill: 4 + + # Decode: 1 worker x TP16 = 16 GPUs = 4 nodes + decode_workers: 1 + decode_nodes: 4 + gpus_per_decode: 16 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + pipeline_parallel_size: 1 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 128 + max_num_tokens: 128 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + - 72 + - 80 + - 88 + - 96 + - 104 + - 112 + - 120 + - 128 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.8 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "2253" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml new file mode 100644 index 0000000000..418aed8e17 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml @@ -0,0 +1,137 @@ +name: "kimi_k25_nvfp4_ISL8K_OSL1K_ctx8dep4_gen1dep32_batch64_eplb0_mtp0" + +# ctx: 8 prefill workers, TP4/EP4 +# gen: 1 decode worker, TP32/EP32, enable_attention_dp=true +# STP (no speculative decoding) +# concurrency: 2151 + +model: + path: "nvidia/Kimi-K2.5-NVFP4" + container: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-dev.1-cuda13" + precision: "fp4" + +resources: + gpu_type: "gb300" + + # Prefill: 8 workers x TP4 = 32 GPUs = 8 nodes + prefill_nodes: 8 + prefill_workers: 8 + gpus_per_prefill: 4 + + # Decode: 1 worker x TP32 = 32 GPUs = 8 nodes + decode_workers: 1 + decode_nodes: 8 + gpus_per_decode: 32 + + gpus_per_node: 4 + +backend: + type: trtllm + + prefill_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + decode_environment: + ENROOT_ALLOW_DEV: "yes" + NCCL_GRAPH_MIXING_SUPPORT: "0" + TLLM_LOG_LEVEL: "INFO" + TRTLLM_ENABLE_PDL: "1" + TRTLLM_SERVER_DISABLE_GC: "1" + TRTLLM_WORKER_DISABLE_GC: "1" + MIMALLOC_PURGE_DELAY: "0" + UCX_TLS: "cuda_ipc,cuda_copy,sm,self,tcp" + + trtllm_config: + prefill: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + enable_attention_dp: true + disable_overlap_scheduler: true + trust_remote_code: true + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + print_iter_log: true + cuda_graph_config: null + moe_config: + backend: CUTEDSL + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + + decode: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + pipeline_parallel_size: 1 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + trust_remote_code: true + max_batch_size: 64 + max_num_tokens: 64 + max_seq_len: 9256 + print_iter_log: true + stream_interval: 100 + num_postprocess_workers: 4 + cuda_graph_config: + enable_padding: true + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + - 24 + - 32 + - 40 + - 48 + - 56 + - 64 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + cache_transceiver_config: + backend: DEFAULT + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + +benchmark: + type: "sa-bench" + isl: 8192 + osl: 1024 + concurrencies: "2151" + req_rate: "inf" + +frontend: + type: "dynamo" + enable_multiple_frontends: true + +health_check: + max_attempts: 360 + interval_seconds: 10 + +dynamo: + request_plane: tcp + install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p6d-dep4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p6d-dep4-tp4.yaml index 7a299d81a5..a541a9975f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p6d-dep4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p6d-dep4-tp4.yaml @@ -112,5 +112,5 @@ benchmark: osl: 1024 concurrencies: "192" req_rate: "inf" - tokenizer_mode: "deepseek_v4" + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" use_chat_template: true diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p9d-tep4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p9d-tep4-tp4.yaml index d1e895bbbf..e5cad82036 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p9d-tep4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p9d-tep4-tp4.yaml @@ -103,5 +103,5 @@ benchmark: osl: 1024 concurrencies: "18" req_rate: "inf" - tokenizer_mode: "deepseek_v4" + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" use_chat_template: true diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml index 0a3643cf27..02adf7d4e6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml @@ -121,5 +121,5 @@ benchmark: osl: 1024 concurrencies: "4096" req_rate: "inf" - tokenizer_mode: "deepseek_v4" + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" use_chat_template: true diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml index 698ef94e2b..50dbfff769 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml @@ -121,5 +121,5 @@ benchmark: osl: 1024 concurrencies: "4096" req_rate: "inf" - tokenizer_mode: "deepseek_v4" + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" use_chat_template: true diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml index 85a1d4c6b6..8e98947683 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml @@ -121,5 +121,5 @@ benchmark: osl: 1024 concurrencies: "4096" req_rate: "inf" - tokenizer_mode: "deepseek_v4" + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" use_chat_template: true diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-7p2d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-7p2d-dep4-dep16.yaml index 53534f9697..602275ba8f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-7p2d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-7p2d-dep4-dep16.yaml @@ -118,5 +118,5 @@ benchmark: osl: 1024 concurrencies: "3072" req_rate: "inf" - tokenizer_mode: "deepseek_v4" + custom_tokenizer: "sa_bench_tokenizers.vllm_deepseek_v4.VLLMDeepseekV4Tokenizer" use_chat_template: true diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb200-2p1d-dep8-dep8-agentic.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb200-2p1d-dep8-dep8-agentic.yaml index b647276179..4ec6ba81d6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb200-2p1d-dep8-dep8-agentic.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb200-2p1d-dep8-dep8-agentic.yaml @@ -148,7 +148,6 @@ srun_options: benchmark: type: custom - aiperf_server_metrics: true command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh env: INFMAX_CONTAINER_WORKSPACE: /infmax-workspace diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb200-3p2d-tep8-tp8-agentic.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb200-3p2d-tep8-tp8-agentic.yaml index 9e2b665843..8405286b88 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb200-3p2d-tep8-tp8-agentic.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb200-3p2d-tep8-tp8-agentic.yaml @@ -142,7 +142,6 @@ srun_options: benchmark: type: custom - aiperf_server_metrics: true command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh env: INFMAX_CONTAINER_WORKSPACE: /infmax-workspace diff --git a/configs/srt-slurm-version.txt b/configs/srt-slurm-version.txt new file mode 100644 index 0000000000..0b8957ae43 --- /dev/null +++ b/configs/srt-slurm-version.txt @@ -0,0 +1 @@ +v1.0.36 diff --git a/runners/launch_b200-dgxc.sh b/runners/launch_b200-dgxc.sh index a276644575..fd38ca8694 100644 --- a/runners/launch_b200-dgxc.sh +++ b/runners/launch_b200-dgxc.sh @@ -6,6 +6,8 @@ SLURM_ACCOUNT="benchmark" set -x +source "$(dirname "${BASH_SOURCE[0]}")/srt_slurm.sh" || exit 1 + # MODEL_PATH: Override with pre-downloaded paths on the shared Lustre tree. # Bench scripts and srt-slurm yaml configs specify HuggingFace model IDs for # portability, but we resolve to /lustre/fsw/models/* here to avoid repeated @@ -96,49 +98,9 @@ if [[ "$IS_MULTINODE" == "true" ]]; then export SERVED_MODEL_NAME=$MODEL - echo "Cloning srt-slurm repository..." SRT_REPO_DIR="srt-slurm" - if [ -d "$SRT_REPO_DIR" ]; then - echo "Removing existing $SRT_REPO_DIR..." - rm -rf "$SRT_REPO_DIR" - fi - - # TODO(CJQ): make first class upon srt-slurm upstream refactor - if [[ "$IS_AGENTIC" == "1" ]]; then - git clone --branch cam/sa-submission-q2-2026 --single-branch https://github.com/cquil11/srt-slurm-nv.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" || exit 1 - elif [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "dsv4" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" || exit 1 - git checkout aflowers/vllm-gb200-v0.20.0 - mkdir -p recipes/vllm/deepseek-v4 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4" recipes/vllm/deepseek-v4 - elif [[ $FRAMEWORK == "dynamo-sglang" && $MODEL_PREFIX == "glm5" && $PRECISION == "fp8" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" || exit 1 - git checkout main - elif [[ $FRAMEWORK == "dynamo-sglang" && $MODEL_PREFIX == "dsr1" && $PRECISION == "fp4" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" || exit 1 - # Pin srt-slurm: newer commits stopped honoring the hash-pinned dynamo - # build and fall back to a dynamo release that is incompatible with this - # sglang image (worker fails at import). This is the last commit before - # that change. Do not float on main -- the srtctl + dynamo-install - # toolchain is unpinned there. - git checkout a98738de9b2233459b5456e9ed71af09ce893f92 - mkdir -p recipes/sglang/dsr1/b200-fp4 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4" recipes/sglang/dsr1/b200-fp4 - elif [[ $FRAMEWORK == "dynamo-trt" && $MODEL_PREFIX == "kimik2.5" && $PRECISION == "fp4" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" || exit 1 - git checkout v1.0.29 - mkdir -p recipes/trtllm/kimi-k25-nvfp4/b200-fp4 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4" recipes/trtllm/kimi-k25-nvfp4/b200-fp4 - else - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" || exit 1 - git checkout sa-submission-q2-2026 - fi + prepare_srt_slurm_checkout "$SRT_REPO_DIR" || exit 1 + cd "$SRT_REPO_DIR" || exit 1 echo "Installing srtctl..." export UV_INSTALL_DIR="$GITHUB_WORKSPACE/.local/bin" diff --git a/runners/launch_b300-nv.sh b/runners/launch_b300-nv.sh index 098e984be5..9a407be59b 100644 --- a/runners/launch_b300-nv.sh +++ b/runners/launch_b300-nv.sh @@ -8,6 +8,8 @@ MINIMAX_M3_SLURM_EXCLUDED_NODELIST="${MINIMAX_M3_SLURM_EXCLUDED_NODELIST-b300-01 set -x +source "$(dirname "${BASH_SOURCE[0]}")/srt_slurm.sh" || exit 1 + if [[ "$IS_MULTINODE" == "true" ]]; then # Validate framework @@ -58,57 +60,13 @@ else exit 1 fi -echo "Cloning srt-slurm repository..." SRT_REPO_DIR="srt-slurm" SRTCTL_SETUP_SCRIPT="" -if [ -d "$SRT_REPO_DIR" ]; then - echo "Removing existing $SRT_REPO_DIR..." - rm -rf "$SRT_REPO_DIR" -fi - -# TODO(CJQ): make first class upon srt-slurm upstream refactor -if [[ "$IS_AGENTIC" == "1" ]]; then - git clone --branch cam/sa-submission-q2-2026 --single-branch https://github.com/cquil11/srt-slurm-nv.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" || exit 1 -elif [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "dsv4" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" || exit 1 - git checkout aflowers/vllm-gb200-v0.20.0 - mkdir -p recipes/vllm/deepseek-v4 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4" recipes/vllm/deepseek-v4 -elif [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "minimaxm3" && $PRECISION == "fp4" && "$CONFIG_FILE" == recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/*.yaml ]]; then - git clone --branch main --single-branch https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" || exit 1 - git checkout c1b6b5c97f323baefad577d70c4e8392b6f537d9 - mkdir -p recipes/vllm/minimax-m3 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3" recipes/vllm/minimax-m3 -elif [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "minimaxm3" && $PRECISION == "fp4" && "$CONFIG_FILE" == recipes/vllm/minimax-m3/b300-fp4/8k1k/*-tp1-*.yaml ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" || exit 1 - git checkout c1fb6989fc5aca803b4ca0f2d17d8be85fad9732 - mkdir -p recipes/vllm/minimax-m3 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3" recipes/vllm/minimax-m3 -elif [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "minimaxm3" && ( $PRECISION == "fp4" || $PRECISION == "fp8" ) ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" || exit 1 - git checkout sa-submission-q2-2026 - mkdir -p recipes/vllm/minimax-m3 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3" recipes/vllm/minimax-m3 - if [[ $PRECISION == "fp8" ]]; then - SRTCTL_SETUP_SCRIPT="minimax-m3-vllm-fixes.sh" - fi - # NVIDIA/srt-slurm#38 - git show 22d46ba9971615016d2339c9ffbc7b4597accfad --format= -- src/srtctl/core/ip_utils/get_node_ip.sh | git apply - || exit 1 - if [[ -n "$SRTCTL_SETUP_SCRIPT" ]]; then - cp \ - "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/configs/$SRTCTL_SETUP_SCRIPT" \ - "configs/$SRTCTL_SETUP_SCRIPT" - fi -else - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" || exit 1 - git checkout sa-submission-q2-2026 +if [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "minimaxm3" && $PRECISION == "fp8" ]]; then + SRTCTL_SETUP_SCRIPT="minimax-m3-vllm-fixes.sh" fi +prepare_srt_slurm_checkout "$SRT_REPO_DIR" || exit 1 +cd "$SRT_REPO_DIR" || exit 1 echo "Installing srtctl..." export UV_INSTALL_DIR="$GITHUB_WORKSPACE/.local/bin" diff --git a/runners/launch_gb200-nv.sh b/runners/launch_gb200-nv.sh index 426cd2dd87..58f69c6214 100755 --- a/runners/launch_gb200-nv.sh +++ b/runners/launch_gb200-nv.sh @@ -5,6 +5,7 @@ set -x source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" +source "$(dirname "${BASH_SOURCE[0]}")/srt_slurm.sh" || exit 1 export SLURM_PARTITION="batch" export SLURM_ACCOUNT="benchmark" @@ -337,7 +338,6 @@ if [[ -z "$CONFIG_FILE" ]]; then exit 1 fi -echo "Cloning srt-slurm repository..." SRT_REPO_DIR="srt-slurm" SRTCTL_SETUP_SCRIPT="" if uses_watchtower_shared_fs; then @@ -345,95 +345,12 @@ if uses_watchtower_shared_fs; then RUN_KEY="${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}-${RUNNER_NAME}-$$" SRT_REPO_DIR="${SHARED_BASE}/srt-slurm-${RUN_KEY}" fi -if [ -d "$SRT_REPO_DIR" ]; then - echo "Removing existing $SRT_REPO_DIR..." - rm -rf "$SRT_REPO_DIR" -fi -# TODO(CJQ): make first class upon srt-slurm upstream refactor -if [[ "$IS_AGENTIC" == "1" ]]; then - # Agentic multi-node uses the same pinned cquil11/srt-slurm-nv commit as - # launch_gb300-nv.sh — everything the agentic recipes need is there: - # - BenchmarkType.CUSTOM + benchmark.command + benchmark.env - # (the hook that hands off to benchmarks/multi_node/agentic_srt.sh) - # - DynamoConfig.wheel (recipes pin the ai-dynamo wheel) - # - srtctl apply --no-preflight (model path /mnt/numa1 is compute-node - # local NVMe, invisible to the login-node runner) - # - benchmark_stage srun_options propagation (container-remap-root - # must reach the agentic_srt.sh srun) - git clone https://github.com/cquil11/srt-slurm-nv.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout de59739b172e507e15ebf145bfe305f606e82fbf - mkdir -p recipes/vllm/deepseek-v4/agentic - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic" \ - recipes/vllm/deepseek-v4/agentic -elif [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "dsv4" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout aflowers/vllm-gb200-v0.20.0 - # Use `cp -rT` so if the upstream branch ever ships a stub - # `recipes/vllm/deepseek-v4/` directory, we overlay our recipes onto - # it rather than nesting (`cp -r src dst` would create - # `recipes/vllm/deepseek-v4/deepseek-v4/...` in that case). - mkdir -p recipes/vllm/deepseek-v4 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4" recipes/vllm/deepseek-v4 -elif [[ $FRAMEWORK == "dynamo-sglang" && $MODEL_PREFIX == "dsv4" ]]; then - # Stay on NVIDIA/srt-slurm:main (default) — submission branch no - # longer needed; overlay our hand-rolled DSV4 sglang recipes onto it. - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - mkdir -p recipes/sglang/deepseek-v4 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4" recipes/sglang/deepseek-v4 -elif [[ $FRAMEWORK == "dynamo-sglang" && $MODEL_PREFIX == "glm5.1" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout sa-submission-q2-2026 - mkdir -p recipes/sglang/glm5 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5" recipes/sglang/glm5 -elif [[ $FRAMEWORK == "dynamo-sglang" && $MODEL_PREFIX == "qwen3.5" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - mkdir -p recipes/sglang/qwen3.5 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5" recipes/sglang/qwen3.5 -elif [[ $FRAMEWORK == "dynamo-sglang" && $MODEL_PREFIX == "glm5.1" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - mkdir -p recipes/sglang/glm5 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5" recipes/sglang/glm5 -elif [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "minimaxm3" && $PRECISION == "fp8" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" || exit 1 - cd "$SRT_REPO_DIR" || exit 1 - git checkout sa-submission-q2-2026 || exit 1 - mkdir -p recipes/vllm/minimax-m3-gb200-fp8 || exit 1 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8" recipes/vllm/minimax-m3-gb200-fp8 || exit 1 +if [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "minimaxm3" && $PRECISION == "fp8" ]]; then SRTCTL_SETUP_SCRIPT="minimax-m3-gb200-vllm-fixes.sh" - cp \ - "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/configs/$SRTCTL_SETUP_SCRIPT" \ - "configs/$SRTCTL_SETUP_SCRIPT" || exit 1 -elif [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "kimik2.5" && $PRECISION == "fp4" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" || exit 1 - cd "$SRT_REPO_DIR" || exit 1 - git checkout main || exit 1 - mkdir -p recipes/vllm/kimi-k2.5-fp4 || exit 1 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4" recipes/vllm/kimi-k2.5-fp4 || exit 1 -elif [[ $FRAMEWORK == "dynamo-vllm" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout sa-submission-q2-2026 -elif [[ $FRAMEWORK == "dynamo-trt" && $MODEL_PREFIX == "kimik2.5" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout sa-submission-q2-2026 -elif [[ $FRAMEWORK == "dynamo-trt" && $MODEL_PREFIX == "glm5" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout v1.0.26 - mkdir -p recipes/trtllm/glm5 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5" recipes/trtllm/glm5 -else - git clone --branch cam/sa-submission-q2-2026 --single-branch https://github.com/cquil11/srt-slurm-nv.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" fi +prepare_srt_slurm_checkout "$SRT_REPO_DIR" || exit 1 +cd "$SRT_REPO_DIR" || exit 1 echo "Installing srtctl..." curl -LsSf https://astral.sh/uv/install.sh | sh diff --git a/runners/launch_gb300-nv.sh b/runners/launch_gb300-nv.sh index e5a8d059b2..d4aad51a79 100644 --- a/runners/launch_gb300-nv.sh +++ b/runners/launch_gb300-nv.sh @@ -4,6 +4,8 @@ set -exo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/srt_slurm.sh" || exit 1 + export SLURM_PARTITION="batch_1" export SLURM_ACCOUNT="benchmark" export ENROOT_ROOTFS_WRITABLE=1 @@ -132,108 +134,19 @@ export EVAL_ONLY="${EVAL_ONLY:-false}" export ISL="$ISL" export OSL="$OSL" -echo "Cloning srt-slurm repository..." RUN_KEY=$(printf "%s" "${RESULT_FILENAME:-${RUNNER_NAME:-gb300-nv}}" | sha1sum | cut -c1-12) SRT_REPO_DIR="${GITHUB_WORKSPACE}/srt-slurm-${GITHUB_RUN_ID:-manual}-${GITHUB_RUN_ATTEMPT:-0}-${RUN_KEY}" SRTCTL_SETUP_SCRIPT="" -rm -rf "$SRT_REPO_DIR" - -if [[ "$IS_AGENTIC" == "1" && $FRAMEWORK == "dynamo-sglang" && $MODEL_PREFIX == "dsv4" ]]; then - # DSv4 GB300 sglang agentic: NVIDIA/srt-slurm v1.0.10 has the nginx - # client_max_body_size fix (>1 MiB agentic warmup bodies), the - # session-affinity frontend, and the BenchmarkType.CUSTOM / extra_mount - # schema these recipes need. - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout v1.0.10 - mkdir -p recipes/sglang/deepseek-v4/agentic - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/agentic" \ - recipes/sglang/deepseek-v4/agentic -elif [[ "$IS_AGENTIC" == "1" ]]; then - # Agentic recipes use NVIDIA/srt-slurm v1.0.36. This is the upstream - # version validated in InferenceX PR #2302 and includes per-node DP, - # matching Dynamo health counts, multi-node TP port handling, and - # Mooncake compatibility. Keep it pinned so sweeps are reproducible. - git clone --branch v1.0.36 --single-branch https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" || exit 1 - cd "$SRT_REPO_DIR" || exit 1 - - mkdir -p recipes/vllm/deepseek-v4/agentic || exit 1 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic" \ - recipes/vllm/deepseek-v4/agentic || exit 1 -elif [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "dsv4" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout aflowers/gb200-dsv4-recipes - mkdir -p recipes/vllm/deepseek-v4 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4" recipes/vllm/deepseek-v4 -elif [[ $FRAMEWORK == "dynamo-sglang" && $MODEL_PREFIX == "glm5" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout main - if [[ $PRECISION == "fp4" ]]; then - mkdir -p recipes/sglang/glm5/gb300-fp4 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4" recipes/sglang/glm5/gb300-fp4 - fi -elif [[ $FRAMEWORK == "dynamo-sglang" && $MODEL_PREFIX == "glm5.1" && $PRECISION == "fp8" ]]; then - # GLM-5.1 FP8 (gb300) recipes are version-controlled in-repo; overlay them - # onto the pinned submission branch. - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout sa-submission-q2-2026 - mkdir -p recipes/sglang/glm5.1 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1" recipes/sglang/glm5.1 -elif [[ $FRAMEWORK == "dynamo-sglang" && $MODEL_PREFIX == "glm5.1" ]]; then - # GLM-5.1 MTP recipe (recipes/gb300-fp4/glm5-mtp.yaml) lives on - # NVIDIA/srt-slurm:main — check it out; no in-repo overlay needed. - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout main -elif [[ $FRAMEWORK == "dynamo-sglang" && $MODEL_PREFIX == "qwen3.5" ]]; then - # Overlay our version-controlled Qwen3.5 recipes onto the srt-slurm checkout. - # fp8 recipes pin dynamo by commit hash (source install), which needs the - # cargo/maturin bootstrap included in the srt-slurm v1.0.25 release — the - # sa-submission-q2-2026 sglang install path assumes maturin ships in the - # image, and the lmsysorg/sglang nightly-dev-cu13 image doesn't include it. - # Same branch the identical gb200-fp8 recipes run on. fp4 recipes pin - # dynamo by version (pip install) and stay on the submission branch they - # were validated against. - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - if [[ $PRECISION == "fp8" ]]; then - git checkout v1.0.25 - else - git checkout sa-submission-q2-2026 - fi - mkdir -p recipes/sglang/qwen3.5 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5" recipes/sglang/qwen3.5 -elif [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "minimaxm3" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout sa-submission-q2-2026 - mkdir -p recipes/vllm/minimax-m3-gb300-fp8 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8" recipes/vllm/minimax-m3-gb300-fp8 +if [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "minimaxm3" ]]; then SRTCTL_SETUP_SCRIPT="minimax-m3-gb300-vllm-fixes.sh" - cp \ - "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/configs/$SRTCTL_SETUP_SCRIPT" \ - "configs/$SRTCTL_SETUP_SCRIPT" -elif [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "kimik2.5" && $PRECISION == "fp4" ]]; then - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout main - mkdir -p recipes/vllm/kimi-k2.5-fp4 - cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4" recipes/vllm/kimi-k2.5-fp4 -elif [[ $FRAMEWORK == "dynamo-trt" && $MODEL_PREFIX == "dsv4" ]]; then +fi +if [[ $FRAMEWORK == "dynamo-trt" && $MODEL_PREFIX == "dsv4" ]]; then # DSv4 dynamo-trt recipes use the HuggingFace model ID as model.path, # so override SRT_SLURM_MODEL_PREFIX to match the recipe's model path key. SRT_SLURM_MODEL_PREFIX="deepseek-ai/DeepSeek-V4-Pro" - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout sa-submission-q2-2026 -else - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout sa-submission-q2-2026 fi +prepare_srt_slurm_checkout "$SRT_REPO_DIR" || exit 1 +cd "$SRT_REPO_DIR" || exit 1 echo "Installing srtctl..." export UV_INSTALL_DIR="$GITHUB_WORKSPACE/.local/bin" @@ -330,9 +243,7 @@ sed -i "s/^name:.*/name: \"${RUNNER_NAME}\"/" "$CONFIG_PATH" # - the agentic path (DSv4-Pro checkpoint), # - glm5.1, whose GLM-5.1-NVFP4 weights are prestaged on the compute-node # /scratch/models, and -# - qwen3.5 fp8, whose weights are also on the compute-node /scratch/models -# and which runs on srt-slurm:v1.0.25 (the release that has the preflight; -# qwen3.5 fp4 runs on sa-submission-q2-2026, which has none). +# - qwen3.5 fp8, whose weights are also on the compute-node /scratch/models. # The engine still fails loudly at runtime if the path is genuinely missing on # the compute node. Other fixed-seq-len recipes resolve model.path to a # login-visible location, so keep the precheck enforced for them. diff --git a/runners/launch_h100-dgxc-slurm.sh b/runners/launch_h100-dgxc-slurm.sh index 1334c95542..793ebc131a 100644 --- a/runners/launch_h100-dgxc-slurm.sh +++ b/runners/launch_h100-dgxc-slurm.sh @@ -11,6 +11,8 @@ SPEC_SUFFIX=$([[ "$SPEC_DECODING" == "mtp" ]] && printf '_mtp' || printf '') set -x +source "$(dirname "${BASH_SOURCE[0]}")/srt_slurm.sh" || exit 1 + if [[ "$IS_MULTINODE" == "true" ]]; then # MODEL_PATH: Override with pre-downloaded paths on H100 runner @@ -38,22 +40,9 @@ if [[ "$IS_MULTINODE" == "true" ]]; then exit 1 fi - echo "Cloning srt-slurm repository..." SRT_REPO_DIR="srt-slurm" - if [ -d "$SRT_REPO_DIR" ]; then - echo "Removing existing $SRT_REPO_DIR..." - rm -rf "$SRT_REPO_DIR" - fi - - # TODO(CJQ): make first class upon srt-slurm upstream refactor - if [[ "$IS_AGENTIC" == "1" ]]; then - git clone --branch cam/sa-submission-q2-2026 --single-branch https://github.com/cquil11/srt-slurm-nv.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - else - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout sa-submission-q2-2026 - fi + prepare_srt_slurm_checkout "$SRT_REPO_DIR" || exit 1 + cd "$SRT_REPO_DIR" || exit 1 echo "Installing srtctl..." export UV_INSTALL_DIR="/mnt/nfs/sa-shared/.uv/bin" diff --git a/runners/launch_h200-dgxc-slurm.sh b/runners/launch_h200-dgxc-slurm.sh index 00c6cc4977..9083029a3e 100755 --- a/runners/launch_h200-dgxc-slurm.sh +++ b/runners/launch_h200-dgxc-slurm.sh @@ -7,6 +7,8 @@ SLURM_ACCOUNT="sa-shared" set -x +source "$(dirname "${BASH_SOURCE[0]}")/srt_slurm.sh" || exit 1 + if [[ "$IS_MULTINODE" == "true" ]]; then # MODEL_PATH: Override with pre-downloaded paths on H200 runner @@ -34,22 +36,9 @@ if [[ "$IS_MULTINODE" == "true" ]]; then exit 1 fi - echo "Cloning srt-slurm repository..." SRT_REPO_DIR="srt-slurm" - if [ -d "$SRT_REPO_DIR" ]; then - echo "Removing existing $SRT_REPO_DIR..." - rm -rf "$SRT_REPO_DIR" - fi - - # TODO(CJQ): make first class upon srt-slurm upstream refactor - if [[ "$IS_AGENTIC" == "1" ]]; then - git clone --branch cam/sa-submission-q2-2026 --single-branch https://github.com/cquil11/srt-slurm-nv.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - else - git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" - cd "$SRT_REPO_DIR" - git checkout sa-submission-q2-2026 - fi + prepare_srt_slurm_checkout "$SRT_REPO_DIR" || exit 1 + cd "$SRT_REPO_DIR" || exit 1 echo "Installing srtctl..." curl -LsSf https://astral.sh/uv/install.sh | sh diff --git a/runners/srt_slurm.sh b/runners/srt_slurm.sh new file mode 100644 index 0000000000..65314a6ba9 --- /dev/null +++ b/runners/srt_slurm.sh @@ -0,0 +1,67 @@ +#!/usr/bin/env bash + +# Shared srt-slurm checkout setup for multinode launchers. + +SRT_SLURM_INFERENCEX_ROOT="$( + cd "$(dirname "${BASH_SOURCE[0]}")/.." >/dev/null 2>&1 || exit 1 + pwd +)" +SRT_SLURM_VERSION_FILE="${SRT_SLURM_INFERENCEX_ROOT}/configs/srt-slurm-version.txt" +SRT_SLURM_REPOSITORY="https://github.com/NVIDIA/srt-slurm.git" + +if [[ ! -r "$SRT_SLURM_VERSION_FILE" ]]; then + echo "Unable to read the srt-slurm version from $SRT_SLURM_VERSION_FILE" >&2 + return 1 +fi + +read -r SRT_SLURM_VERSION < "$SRT_SLURM_VERSION_FILE" +if [[ -z "$SRT_SLURM_VERSION" ]]; then + echo "The srt-slurm version file is empty: $SRT_SLURM_VERSION_FILE" >&2 + return 1 +fi + +prepare_srt_slurm_checkout() { + local destination=$1 + local overlay_root="${SRT_SLURM_INFERENCEX_ROOT}/benchmarks/multi_node/srt-slurm-recipes" + local source_dir + local source_name + local target_dir + + if [[ -z "$destination" || "$destination" == "/" ]]; then + echo "Refusing to prepare an invalid srt-slurm destination: '$destination'" >&2 + return 1 + fi + + echo "Cloning srt-slurm ${SRT_SLURM_VERSION}..." + rm -rf "$destination" + git clone \ + --branch "$SRT_SLURM_VERSION" \ + --single-branch \ + "$SRT_SLURM_REPOSITORY" \ + "$destination" || return 1 + + # Overlay all InferenceX-owned recipes and setup scripts. Keeping these in + # InferenceX lets every launcher use the same tagged srt-slurm code while + # preserving configurations that have not been released upstream. + for source_dir in "$overlay_root"/*; do + [[ -d "$source_dir" ]] || continue + source_name=${source_dir##*/} + if [[ "$source_name" == "configs" ]]; then + target_dir="$destination/configs" + else + target_dir="$destination/recipes/$source_name" + fi + mkdir -p "$target_dir" + cp -R "$source_dir/." "$target_dir/" || return 1 + done + + # The B200 TRT-LLM Kimi K2.5 configs predate the upstream recipe naming + # convention used by CONFIG_FILE, so expose the checked-in recipes at the + # expected path without maintaining a second copy. + source_dir="$overlay_root/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4" + if [[ -d "$source_dir" ]]; then + target_dir="$destination/recipes/trtllm/kimi-k25-nvfp4/b200-fp4" + mkdir -p "$target_dir" + cp -R "$source_dir/." "$target_dir/" || return 1 + fi +} From d8bdc71726b1ad30bcc81cf5405ff25218deac48 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Tue, 28 Jul 2026 10:46:29 -0500 Subject: [PATCH 2/4] refactor: run SRT throughput with InferenceX benchmark MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Route srt-slurm SA-Bench recipes through the custom runner and InferenceX benchmark_serving without checking in converted recipe copies. Preserve recipe concurrency, request-rate, warmup, tokenizer, topology, and result naming settings. 中文:通过 custom runner 和 InferenceX benchmark_serving 运行 srt-slurm 吞吐量配置,无需提交转换后的重复 recipe;同时保留并发、请求速率、预热、分词器、拓扑及结果命名设置。 --- benchmarks/benchmark_lib.sh | 33 ++- benchmarks/multi_node/srt_bench.sh | 80 +++++++ runners/launch_b200-dgxc.sh | 3 + runners/launch_b300-nv.sh | 15 +- runners/launch_gb200-nv.sh | 6 +- runners/launch_gb300-nv.sh | 3 + runners/launch_h100-dgxc-slurm.sh | 3 + runners/launch_h200-dgxc-slurm.sh | 3 + runners/srt_slurm.sh | 8 + runners/srt_slurm_benchmark_config.py | 237 ++++++++++++++++++++ utils/bench_serving/backend_request_func.py | 23 +- 11 files changed, 396 insertions(+), 18 deletions(-) create mode 100755 benchmarks/multi_node/srt_bench.sh create mode 100755 runners/srt_slurm_benchmark_config.py diff --git a/benchmarks/benchmark_lib.sh b/benchmarks/benchmark_lib.sh index aa9cc6304f..28bd28b84e 100644 --- a/benchmarks/benchmark_lib.sh +++ b/benchmarks/benchmark_lib.sh @@ -327,7 +327,7 @@ wait_for_server_ready() { } # Run benchmark serving with standardized parameters -# All parameters are required except --endpoint, --use-chat-template, --dsv4, and --trust-remote-code +# All parameters are required except the optional parameters listed below. # Parameters: # --model: Model name # --port: Server port @@ -346,9 +346,14 @@ wait_for_server_ready() { # template. Implies --use-chat-template. # --trust-remote-code: Optional flag to trust remote code from HuggingFace # --server-pid: Optional server process ID to monitor during benchmark +# --tokenizer: Optional tokenizer name/path override +# --tokenizer-mode: Optional tokenizer implementation mode +# --host: Optional server host (default: 0.0.0.0) +# --request-rate: Optional request rate (default: inf) +# --num-warmups: Optional warmup request count (default: 2 * concurrency) run_benchmark_serving() { # In eval-only mode, skip the throughput benchmark entirely. - if [ "${EVAL_ONLY}" = "true" ]; then + if [ "${EVAL_ONLY:-false}" = "true" ]; then echo "EVAL_ONLY mode: skipping throughput benchmark" return 0 fi @@ -372,6 +377,9 @@ run_benchmark_serving() { local server_pid="" local tokenizer="" local tokenizer_mode="" + local host="0.0.0.0" + local request_rate="inf" + local num_warmups="" while [[ $# -gt 0 ]]; do case $1 in @@ -448,6 +456,18 @@ run_benchmark_serving() { tokenizer_mode="$2" shift 2 ;; + --host) + host="$2" + shift 2 + ;; + --request-rate) + request_rate="$2" + shift 2 + ;; + --num-warmups) + num_warmups="$2" + shift 2 + ;; *) echo "Unknown parameter: $1" return 1 @@ -500,6 +520,9 @@ run_benchmark_serving() { if [[ -z "$workspace_dir" ]]; then workspace_dir=$(pwd) fi + if [[ -z "$num_warmups" ]]; then + num_warmups="$((2 * max_concurrency))" + fi # Profiling support: when PROFILE=1, ensure profiler dir exists, add --profile flag, # and cap num_prompts to keep traces small. @@ -518,18 +541,18 @@ run_benchmark_serving() { python3 "$workspace_dir/utils/bench_serving/benchmark_serving.py" --model "$model" --backend "$backend" - --base-url "http://0.0.0.0:$port" + --base-url "http://${host}:$port" --dataset-name random --random-input-len "$input_len" --random-output-len "$output_len" --random-range-ratio "$random_range_ratio" --num-prompts "$num_prompts" --max-concurrency "$max_concurrency" - --request-rate inf + --request-rate "$request_rate" --ignore-eos "${profile_flag[@]}" --save-result - --num-warmups "$((2 * max_concurrency))" \ + --num-warmups "$num_warmups" \ --percentile-metrics 'ttft,tpot,itl,e2el' --result-dir "$result_dir" --result-filename "$result_filename.json" diff --git a/benchmarks/multi_node/srt_bench.sh b/benchmarks/multi_node/srt_bench.sh new file mode 100755 index 0000000000..63d7db3a5b --- /dev/null +++ b/benchmarks/multi_node/srt_bench.sh @@ -0,0 +1,80 @@ +#!/usr/bin/env bash + +# InferenceX benchmark_serving adapter for srt-slurm's custom benchmark runner. +# The recipe converter in runners/srt_slurm_benchmark_config.py supplies the +# environment below from the selected recipe's SA-Bench fields. + +set -euo pipefail + +INFMAX_WS="${INFMAX_CONTAINER_WORKSPACE:-/infmax-workspace}" +# shellcheck disable=SC1091 +source "$INFMAX_WS/benchmarks/benchmark_lib.sh" + +check_env_vars MODEL TOKENIZER ISL OSL CONC_LIST REQ_RATE DISAGG \ + PREFILL_GPUS DECODE_GPUS TOTAL_GPUS + +BENCH_HOST="${SRT_FRONTEND_HOST:-127.0.0.1}" +BENCH_PORT="${SRT_FRONTEND_PORT:-8000}" +RANDOM_RANGE_RATIO="${RANDOM_RANGE_RATIO:-0.8}" +NUM_PROMPTS_MULT="${NUM_PROMPTS_MULT:-10}" +NUM_WARMUP_MULT="${NUM_WARMUP_MULT:-2}" +USE_CHAT_TEMPLATE="${USE_CHAT_TEMPLATE:-true}" +DSV4="${DSV4:-false}" +TRUST_REMOTE_CODE="${TRUST_REMOTE_CODE:-true}" +RESULT_DIR="/logs/sa-bench_isl_${ISL}_osl_${OSL}" + +ensure_bench_serving_deps() { + local deps=(aiohttp numpy tqdm transformers huggingface_hub) + if python3 -c \ + "import aiohttp, numpy, tqdm, transformers, huggingface_hub" \ + 2>/dev/null; then + return + fi + + local venv="/tmp/inferencex-srt-bench-venv" + [[ -d "$venv" ]] || python3 -m venv --system-site-packages "$venv" + # shellcheck disable=SC1091 + source "$venv/bin/activate" + pip install --quiet "${deps[@]}" +} + +ensure_bench_serving_deps +mkdir -p "$RESULT_DIR" + +curl -fsS "http://${BENCH_HOST}:${BENCH_PORT}/v1/models" >/dev/null || { + echo "InferenceX benchmark could not reach the srt-slurm frontend" >&2 + exit 66 +} +ulimit -n 65536 2>/dev/null || true + +read -r -a concurrencies <<< "${CONC_LIST//x/ }" +for concurrency in "${concurrencies[@]}"; do + if [[ "$DISAGG" == "true" ]]; then + result_filename="results_concurrency_${concurrency}_gpus_${TOTAL_GPUS}_ctx_${PREFILL_GPUS}_gen_${DECODE_GPUS}" + else + result_filename="results_concurrency_${concurrency}_gpus_${TOTAL_GPUS}" + fi + + args=( + --model "$MODEL" + --tokenizer "$TOKENIZER" + --host "$BENCH_HOST" + --port "$BENCH_PORT" + --backend openai + --input-len "$ISL" + --output-len "$OSL" + --random-range-ratio "$RANDOM_RANGE_RATIO" + --num-prompts "$((concurrency * NUM_PROMPTS_MULT))" + --num-warmups "$((concurrency * NUM_WARMUP_MULT))" + --request-rate "$REQ_RATE" + --max-concurrency "$concurrency" + --result-filename "$result_filename" + --result-dir "$RESULT_DIR" + --bench-serving-dir "$INFMAX_WS" + ) + [[ "$USE_CHAT_TEMPLATE" == "true" ]] && args+=(--use-chat-template) + [[ "$DSV4" == "true" ]] && args+=(--dsv4) + [[ "$TRUST_REMOTE_CODE" == "true" ]] && args+=(--trust-remote-code) + + run_benchmark_serving "${args[@]}" +done diff --git a/runners/launch_b200-dgxc.sh b/runners/launch_b200-dgxc.sh index fd38ca8694..721a6fe4d1 100644 --- a/runners/launch_b200-dgxc.sh +++ b/runners/launch_b200-dgxc.sh @@ -223,6 +223,9 @@ EOF # so large-model loads (e.g. DSR1-FP8 ~680GB off shared FS) finish in time. # Uses ${CONFIG_FILE%%:*} because CONFIG_FILE may carry an :override[N] suffix. sed -i 's/^ max_attempts: [0-9]*/ max_attempts: 720/' "${CONFIG_FILE%%:*}" + CONFIG_FILE="$( + prepare_inferencex_srt_benchmark_config "$CONFIG_FILE" + )" || exit 1 SRTCTL_OUTPUT=$(srtctl apply -f "$CONFIG_FILE" --tags "b200,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" 2>&1) echo "$SRTCTL_OUTPUT" diff --git a/runners/launch_b300-nv.sh b/runners/launch_b300-nv.sh index 9a407be59b..d9d2e69802 100644 --- a/runners/launch_b300-nv.sh +++ b/runners/launch_b300-nv.sh @@ -147,18 +147,23 @@ sed -i "s/^name:.*/name: \"${RUNNER_NAME}\"/" "$CONFIG_FILE" if [[ "$MODEL_PREFIX" == "minimaxm3" && -n "$MINIMAX_M3_SLURM_EXCLUDED_NODELIST" ]]; then sed -i "/^name:.*/a sbatch_directives:\n exclude: \"${MINIMAX_M3_SLURM_EXCLUDED_NODELIST}\"" "$CONFIG_FILE" fi -SRTCTL_APPLY_ARGS=( - -f "$CONFIG_FILE" - --tags "b300,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" -) # The MTP and TP1 8k1k recipes use newer srt-slurm revisions whose preflight checks # model.path on this GHA login host. MiniMax-M3 NVFP4 is intentionally staged # under compute-node-local /scratch (as in the original B300 submission), so # the login host cannot stat it even though workers can. Keep this bypass # scoped to those recipe sets; runtime model loading still validates the path. +PREFLIGHT_ARGS=() if [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "minimaxm3" && $PRECISION == "fp4" && ( "$CONFIG_FILE" == recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/*.yaml || "$CONFIG_FILE" == recipes/vllm/minimax-m3/b300-fp4/8k1k/*-tp1-*.yaml ) ]]; then - SRTCTL_APPLY_ARGS+=(--no-preflight) + PREFLIGHT_ARGS=(--no-preflight) fi +CONFIG_FILE="$( + prepare_inferencex_srt_benchmark_config "$CONFIG_FILE" +)" || exit 1 +SRTCTL_APPLY_ARGS=( + "${PREFLIGHT_ARGS[@]}" + -f "$CONFIG_FILE" + --tags "b300,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" +) if [[ -n "$SRTCTL_SETUP_SCRIPT" ]]; then SRTCTL_APPLY_ARGS+=(--setup-script "$SRTCTL_SETUP_SCRIPT") fi diff --git a/runners/launch_gb200-nv.sh b/runners/launch_gb200-nv.sh index 58f69c6214..48e0768496 100755 --- a/runners/launch_gb200-nv.sh +++ b/runners/launch_gb200-nv.sh @@ -495,11 +495,11 @@ if [[ "$IS_AGENTIC" == "1" ]]; then PREFLIGHT_ARGS=(--no-preflight) fi +CONFIG_FILE="$( + prepare_inferencex_srt_benchmark_config "$CONFIG_FILE" +)" || exit 1 SRTCTL_APPLY_ARGS=( "${PREFLIGHT_ARGS[@]}" - # Pass the full CONFIG_FILE (not the stripped CONFIG_PATH): srtctl needs the - # ":zip_override_...[i]" selector to pick the recipe block. For plain-file - # recipes CONFIG_FILE == CONFIG_PATH, so this is a no-op for them. -f "$CONFIG_FILE" --tags "gb200,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" ) diff --git a/runners/launch_gb300-nv.sh b/runners/launch_gb300-nv.sh index d4aad51a79..9798df5857 100644 --- a/runners/launch_gb300-nv.sh +++ b/runners/launch_gb300-nv.sh @@ -247,6 +247,9 @@ sed -i "s/^name:.*/name: \"${RUNNER_NAME}\"/" "$CONFIG_PATH" # The engine still fails loudly at runtime if the path is genuinely missing on # the compute node. Other fixed-seq-len recipes resolve model.path to a # login-visible location, so keep the precheck enforced for them. +CONFIG_FILE="$( + prepare_inferencex_srt_benchmark_config "$CONFIG_FILE" +)" || exit 1 SRTCTL_APPLY_ARGS=( -f "$CONFIG_FILE" --tags "gb300,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" diff --git a/runners/launch_h100-dgxc-slurm.sh b/runners/launch_h100-dgxc-slurm.sh index 793ebc131a..42642ab444 100644 --- a/runners/launch_h100-dgxc-slurm.sh +++ b/runners/launch_h100-dgxc-slurm.sh @@ -134,6 +134,9 @@ EOF sed -i "s/^name:.*/name: \"${RUNNER_NAME}\"/" "$CONFIG_FILE" # Raise sglang's torch-distributed TCPStore timeout from the 600s gloo default sed -i '/^ watchdog-timeout:/a\ dist-timeout: 1800' "${CONFIG_FILE%%:*}" + CONFIG_FILE="$( + prepare_inferencex_srt_benchmark_config "$CONFIG_FILE" + )" || exit 1 SRTCTL_OUTPUT=$(srtctl apply -f "$CONFIG_FILE" --tags "h100,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" 2>&1) echo "$SRTCTL_OUTPUT" diff --git a/runners/launch_h200-dgxc-slurm.sh b/runners/launch_h200-dgxc-slurm.sh index 9083029a3e..28cdd21f73 100755 --- a/runners/launch_h200-dgxc-slurm.sh +++ b/runners/launch_h200-dgxc-slurm.sh @@ -124,6 +124,9 @@ EOF sed -i "s/^name:.*/name: \"${RUNNER_NAME}\"/" "$CONFIG_FILE" sed -i '/^health_check:/,/^[^ ]/{ /^health_check:/d; /^ /d; }' "${CONFIG_FILE%%:*}" printf '\nhealth_check:\n max_attempts: 720\n interval_seconds: 10\n' >> "${CONFIG_FILE%%:*}" + CONFIG_FILE="$( + prepare_inferencex_srt_benchmark_config "$CONFIG_FILE" + )" || exit 1 SRTCTL_OUTPUT=$(srtctl apply -f "$CONFIG_FILE" --tags "h200,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" 2>&1) echo "$SRTCTL_OUTPUT" diff --git a/runners/srt_slurm.sh b/runners/srt_slurm.sh index 65314a6ba9..8eb2bb55ea 100644 --- a/runners/srt_slurm.sh +++ b/runners/srt_slurm.sh @@ -65,3 +65,11 @@ prepare_srt_slurm_checkout() { cp -R "$source_dir/." "$target_dir/" || return 1 fi } + +prepare_inferencex_srt_benchmark_config() { + local config_spec=$1 + + python \ + "${SRT_SLURM_INFERENCEX_ROOT}/runners/srt_slurm_benchmark_config.py" \ + "$config_spec" +} diff --git a/runners/srt_slurm_benchmark_config.py b/runners/srt_slurm_benchmark_config.py new file mode 100755 index 0000000000..0f1e6deb7c --- /dev/null +++ b/runners/srt_slurm_benchmark_config.py @@ -0,0 +1,237 @@ +#!/usr/bin/env python3 +"""Render an srt-slurm SA-Bench recipe for InferenceX's custom benchmark.""" + +from __future__ import annotations + +import argparse +import hashlib +from pathlib import Path +from typing import Any + +import yaml + +from srtctl.core.config import ( + load_cluster_config, + resolve_config_with_defaults, + resolve_override_yaml, +) +from srtctl.core.schema import SrtConfig + + +CUSTOM_BENCHMARK_COMMAND = ( + "bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh" +) + + +def split_config_spec(config_spec: str) -> tuple[Path, str | None]: + """Split srtctl's ``path.yaml:selector`` syntax.""" + marker = ".yaml:" + if marker not in config_spec: + return Path(config_spec), None + path, selector = config_spec.split(marker, 1) + return Path(f"{path}.yaml"), selector + + +def to_plain_data(value: Any) -> Any: + """Remove ruamel container and scalar types before PyYAML serialization.""" + if isinstance(value, dict): + return {key: to_plain_data(item) for key, item in value.items()} + if isinstance(value, list): + return [to_plain_data(item) for item in value] + if isinstance(value, str): + return str(value) + if isinstance(value, bool): + return bool(value) + if isinstance(value, int): + return int(value) + if isinstance(value, float): + return float(value) + return value + + +def load_selected_recipes( + config_spec: str, +) -> tuple[Path, list[tuple[str | None, dict[str, Any]]], bool]: + """Load a plain recipe or resolve the requested override variants.""" + config_path, selector = split_config_spec(config_spec) + with config_path.open(encoding="utf-8") as config_file: + raw_config = yaml.safe_load(config_file) + + if not isinstance(raw_config, dict): + raise ValueError(f"{config_path} is not a YAML mapping") + + if "base" not in raw_config: + if selector is not None: + raise ValueError( + f"{config_path} is not an override recipe, but selector " + f"{selector!r} was provided" + ) + return config_path, [(None, raw_config)], False + + if selector is None: + variants = [ + (suffix, to_plain_data(recipe)) + for suffix, recipe in resolve_override_yaml(config_path) + ] + return config_path, variants, True + + variants = resolve_override_yaml(config_path, selector) + if not variants: + raise ValueError( + f"{config_spec} did not resolve to any variants" + ) + return ( + config_path, + [(None, to_plain_data(recipe)) for _, recipe in variants], + len(variants) > 1, + ) + + +def stringify_concurrencies(concurrencies: list[int] | str | None) -> str: + """Return the whitespace-separated format consumed by srt_bench.sh.""" + if isinstance(concurrencies, list): + return " ".join(str(value) for value in concurrencies) + if concurrencies is None: + return "" + return str(concurrencies).replace("x", " ") + + +def custom_benchmark_environment(config: SrtConfig) -> dict[str, str]: + """Translate SA-Bench's typed settings to the custom wrapper contract.""" + benchmark = config.benchmark + resources = config.resources + + if benchmark.dataset_name not in (None, "random"): + raise ValueError( + "InferenceX benchmark_serving currently supports only the random " + f"dataset, not {benchmark.dataset_name!r}" + ) + if benchmark.reuse_http_connections: + raise ValueError( + "InferenceX benchmark_serving does not support " + "benchmark.reuse_http_connections" + ) + if ( + benchmark.slow_down_sleep_time is not None + or benchmark.slow_down_wait_time is not None + ): + raise ValueError( + "InferenceX benchmark_serving does not support SA-Bench slow_down" + ) + + concurrencies = stringify_concurrencies(benchmark.concurrencies) + if not concurrencies: + raise ValueError("benchmark.concurrencies is required") + if benchmark.isl is None or benchmark.osl is None: + raise ValueError("benchmark.isl and benchmark.osl are required") + + if resources.is_disaggregated: + prefill_gpus = resources.prefill_gpus + decode_gpus = resources.decode_gpus + total_gpus = prefill_gpus + decode_gpus + else: + prefill_gpus = 0 + decode_gpus = 0 + total_gpus = (resources.agg_nodes or 1) * resources.gpus_per_node + + model_path = config.model.path + tokenizer = ( + model_path.removeprefix("hf:") + if model_path.startswith("hf:") + else "/model" + ) + custom_tokenizer = benchmark.custom_tokenizer or "" + is_deepseek_v4 = "deepseek_v4" in custom_tokenizer.lower() + + return { + "MODEL": config.served_model_name, + "TOKENIZER": tokenizer, + "ISL": str(benchmark.isl), + "OSL": str(benchmark.osl), + "CONC_LIST": concurrencies, + "REQ_RATE": str(benchmark.req_rate or "inf"), + "DISAGG": str(resources.is_disaggregated).lower(), + "PREFILL_GPUS": str(prefill_gpus), + "DECODE_GPUS": str(decode_gpus), + "TOTAL_GPUS": str(total_gpus), + "RANDOM_RANGE_RATIO": str(benchmark.random_range_ratio or 0.8), + "NUM_PROMPTS_MULT": str(benchmark.num_prompts_mult or 10), + "NUM_WARMUP_MULT": str( + benchmark.num_warmup_mult + if benchmark.num_warmup_mult is not None + else 2 + ), + "USE_CHAT_TEMPLATE": str(benchmark.use_chat_template).lower(), + "DSV4": str( + is_deepseek_v4 and benchmark.use_chat_template + ).lower(), + "TRUST_REMOTE_CODE": "true", + } + + +def convert_recipe(recipe: dict[str, Any]) -> bool: + """Convert an SA-Bench recipe in place; return whether it changed.""" + benchmark = recipe.get("benchmark") + if not isinstance(benchmark, dict) or benchmark.get("type") != "sa-bench": + return False + + cluster_config = load_cluster_config() + resolved = resolve_config_with_defaults(recipe, cluster_config) + config = SrtConfig.Schema().load(resolved) + if not isinstance(config, SrtConfig): + raise TypeError("srt-slurm did not return a typed SrtConfig") + + environment = dict(benchmark.get("env") or {}) + environment.update(custom_benchmark_environment(config)) + benchmark["type"] = "custom" + benchmark["command"] = CUSTOM_BENCHMARK_COMMAND + benchmark["env"] = environment + return True + + +def render_config(config_spec: str, output_dir: Path) -> str: + """Render selected recipes and return their path for ``srtctl apply``.""" + config_path, variants, keep_override_format = load_selected_recipes( + config_spec + ) + changed = False + for _, recipe in variants: + changed = convert_recipe(recipe) or changed + if not changed: + return config_spec + + if keep_override_format: + rendered: dict[str, Any] = {"base": {}} + for index, (suffix, recipe) in enumerate(variants): + variant_name = suffix or f"inferencex_{index}" + rendered[f"override_{variant_name}"] = recipe + else: + if len(variants) != 1: + raise ValueError( + f"{config_spec} resolved to {len(variants)} variants but " + "cannot preserve their selector" + ) + rendered = variants[0][1] + + digest = hashlib.sha256(config_spec.encode()).hexdigest()[:12] + output_dir.mkdir(parents=True, exist_ok=True) + output_path = output_dir / f"{config_path.stem}-{digest}.yaml" + with output_path.open("w", encoding="utf-8") as output_file: + yaml.safe_dump(rendered, output_file, sort_keys=False) + return str(output_path) + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("config_spec") + parser.add_argument( + "--output-dir", + type=Path, + default=Path(".inferencex-recipes"), + ) + args = parser.parse_args() + print(render_config(args.config_spec, args.output_dir)) + + +if __name__ == "__main__": + main() diff --git a/utils/bench_serving/backend_request_func.py b/utils/bench_serving/backend_request_func.py index 4c8820f8da..d2e912d14c 100644 --- a/utils/bench_serving/backend_request_func.py +++ b/utils/bench_serving/backend_request_func.py @@ -553,11 +553,24 @@ def get_tokenizer( return MistralTokenizer.from_pretrained( str(pretrained_model_name_or_path)) else: - tokenizer = AutoTokenizer.from_pretrained( - pretrained_model_name_or_path, - trust_remote_code=trust_remote_code, - **kwargs, - ) + # DeepSeek-V4 ships tokenizer.json but its model_type is not + # registered by stock Transformers. Load that fast tokenizer directly + # so InferenceX's self-contained --dsv4 prompt encoder can be used + # without srt-slurm's separate sa_bench_tokenizers package. + try: + tokenizer = AutoTokenizer.from_pretrained( + pretrained_model_name_or_path, + trust_remote_code=trust_remote_code, + **kwargs, + ) + except ValueError as error: + if "deepseek_v4" not in str(error).lower(): + raise + tokenizer = PreTrainedTokenizerFast.from_pretrained( + pretrained_model_name_or_path, + trust_remote_code=trust_remote_code, + **kwargs, + ) return _fix_tokenizer_for_sglang(tokenizer, pretrained_model_name_or_path) From 1bd966976fa808ad2d2aeeb4fa4da34875bf671d Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Tue, 28 Jul 2026 11:07:43 -0500 Subject: [PATCH 3/4] feat: use custom benchmark in all SRT recipes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将所有 SRT 配方统一改为 InferenceX 自定义基准测试,并由共享渲染器补全运行环境。 --- ...10dep4_gen1dep16_batch64_eplb384_mtp3.yaml | 3 ++- ...12dep4_gen1dep8_batch512_eplb384_mtp1.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml | 3 ++- ...tx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml | 3 ++- ...tx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml | 3 ++- ...tx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml | 3 ++- ...x6dep4_gen1dep16_batch32_eplb384_mtp3.yaml | 3 ++- ...x7dep4_gen1dep8_batch128_eplb384_mtp3.yaml | 3 ++- ...x8dep4_gen1dep32_batch16_eplb384_mtp3.yaml | 3 ++- ...x9dep4_gen1dep8_batch256_eplb384_mtp1.yaml | 3 ++- ...10dep4_gen1dep8_batch512_eplb384_mtp0.yaml | 3 ++- ...tx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml | 3 ++- ...tx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml | 3 ++- ...x4dep4_gen1dep32_batch16_eplb384_mtp0.yaml | 3 ++- ...x5dep4_gen1dep16_batch64_eplb384_mtp0.yaml | 3 ++- ...x6dep4_gen1dep32_batch32_eplb384_mtp0.yaml | 3 ++- ...x6dep4_gen1dep8_batch256_eplb384_mtp0.yaml | 3 ++- ...9dep4_gen1dep16_batch128_eplb384_mtp0.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml | 3 ++- .../ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml | 3 ++- ...ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml | 3 ++- ...ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml | 3 ++- ...ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml | 3 ++- ...tx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml | 3 ++- ...tx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml | 3 ++- .../ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml | 3 ++- ...ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml | 3 ++- ...ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml | 3 ++- ...tx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml | 3 ++- ...ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml | 3 ++- .../disagg-gb200-1p1d-dep8-dep16-6-c512.yaml | 3 ++- .../8k1k/disagg-gb200-1p1d-tp8-tp8-4-c1.yaml | 3 ++- .../disagg-gb200-1p2d-dep8-dep16-10-c256.yaml | 3 ++- .../disagg-gb200-1p4d-dep8-tp8-10-c64.yaml | 3 ++- .../disagg-gb200-2p1d-dep8-dep16-8-c1536.yaml | 3 ++- ...disagg-gb200-4p1d-dep8-dep16-12-c4096.yaml | 3 ++- ...disagg-gb200-5p1d-dep8-dep16-14-c8192.yaml | 3 ++- ...disagg-gb200-6p1d-dep8-dep12-15-c8192.yaml | 3 ++- ...gg-gb200-low-latency-1p1d-tp8-tp8-mtp.yaml | 3 ++- ...g-gb200-low-latency-1p6d-dep8-tp8-mtp.yaml | 3 ++- ...g-gb200-mid-curve-1p1d-dep8-dep16-mtp.yaml | 3 ++- ...g-gb200-mid-curve-2p1d-dep8-dep16-mtp.yaml | 3 ++- ...g-gb200-mid-curve-3p1d-dep8-dep16-mtp.yaml | 3 ++- ...g-gb200-mid-curve-4p1d-dep8-dep16-mtp.yaml | 3 ++- ...g-gb200-mid-curve-5p1d-dep8-dep16-mtp.yaml | 3 ++- ...g-gb200-mid-curve-6p1d-dep8-dep16-mtp.yaml | 3 ++- ...isagg-gb300-10p1d-dep4-dep32-18-c2500.yaml | 3 ++- ...isagg-gb300-12p1d-dep4-dep24-18-c3000.yaml | 3 ++- ...isagg-gb300-14p1d-dep4-dep16-18-c8192.yaml | 3 ++- ...sagg-gb300-15p1d-dep4-dep12-18-c12000.yaml | 3 ++- .../disagg-gb300-1p1d-dep4-dep16-5-c1024.yaml | 3 ++- .../8k1k/disagg-gb300-1p1d-tp4-tp4-2-c1.yaml | 3 ++- ...disagg-gb300-8p1d-dep4-dep40-18-c2048.yaml | 3 ++- .../disagg-high-conc-6p1d-dep4-dep8-mtp.yaml | 3 ++- .../disagg-high-conc-8p1d-dep4-dep8-mtp.yaml | 3 ++- .../disagg-low-latency-1p1d-tp4-tp4-mtp.yaml | 3 ++- .../disagg-low-latency-1p6d-dep4-tp4-mtp.yaml | 3 ++- .../disagg-mid-curve-1p1d-dep4-dep16-mtp.yaml | 3 ++- .../disagg-mid-curve-1p1d-dep4-dep8-mtp.yaml | 3 ++- .../disagg-mid-curve-2p1d-dep4-dep8-mtp.yaml | 3 ++- .../disagg-mid-curve-4p1d-dep4-dep8-mtp.yaml | 3 ++- .../1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml | 3 ++- .../1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml | 3 ++- .../1k1k/disagg/mtp/1k1k_mtp_maxtpt_0.yaml | 3 ++- .../1k1k/disagg/mtp/1k1k_mtp_maxtpt_1.yaml | 3 ++- .../disagg/mtp/8k1k_mtp2_throughput_1p1d.yaml | 3 ++- .../disagg/mtp/8k1k_mtp2_throughput_2p1d.yaml | 3 ++- .../disagg/mtp/8k1k_mtp2_throughput_3p1d.yaml | 3 ++- .../disagg/mtp/8k1k_mtp2_throughput_4p1d.yaml | 3 ++- .../disagg/mtp/8k1k_mtp2_throughput_5p1d.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_4p1d_c2048.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml | 3 ++- .../1k1k/disagg/mtp/1k1k_mtp_hightpt_3.yaml | 3 ++- .../1k1k/disagg/mtp/1k1k_mtp_hightpt_4.yaml | 3 ++- .../1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml | 3 ++- .../1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_hightpt_0.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_hightpt_1.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_hightpt_2.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_hightpt_3.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_maxtpt_0.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_5.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_maxtpt_3.yaml | 3 ++- .../sglang/glm5/gb200-fp4/glm5-mtp.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_hightpt_0.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml | 3 ++- .../1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_hightpt_0.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_hightpt_1.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_hightpt_2.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_hightpt_3.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml | 3 ++- .../sglang/glm5/gb200-fp8/glm5-mtp.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml | 3 ++- .../gb200-fp8/1k1k/1p1d-dep4-dep16.yaml | 3 ++- .../qwen3.5/gb200-fp8/1k1k/1p1d-tp4-tp4.yaml | 3 ++- .../gb200-fp8/1k1k/2p1d-dep4-dep16.yaml | 3 ++- .../qwen3.5/gb200-fp8/8k1k/1p1d-tp4-tp4.yaml | 3 ++- .../gb200-fp8/8k1k/4p1d-dep4-dep16.yaml | 3 ++- .../gb200-fp8/8k1k/8p1d-dep4-dep16.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_maxtpt_0.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_maxtpt_1.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_maxtpt_2.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_maxtpt_3.yaml | 3 ++- .../8k1k/disagg/mtp/8k1k_mtp_maxtpt_4.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml | 5 ++-- .../8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml | 3 ++- .../8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml | 3 ++- .../gb300-fp8/1k1k/1p1d-dep4-dep16.yaml | 3 ++- .../qwen3.5/gb300-fp8/1k1k/1p1d-tp4-tp4.yaml | 3 ++- .../gb300-fp8/1k1k/2p1d-dep4-dep16.yaml | 3 ++- .../qwen3.5/gb300-fp8/8k1k/1p1d-tp4-tp4.yaml | 3 ++- .../gb300-fp8/8k1k/4p1d-dep4-dep16.yaml | 3 ++- .../gb300-fp8/8k1k/8p1d-dep4-dep16.yaml | 3 ++- ...ctx1dep4_gen1dep16_batch32_eplb0_mtp3.yaml | 3 ++- ...ctx1dep4_gen1dep16_batch64_eplb0_mtp1.yaml | 3 ++- ...ctx1dep4_gen1dep32_batch16_eplb0_mtp3.yaml | 3 ++- .../ctx1dep4_gen1dep32_batch4_eplb0_mtp3.yaml | 3 ++- .../ctx1dep4_gen1dep32_batch8_eplb0_mtp3.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch16_eplb0_mtp3.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch4_eplb0_mtp3.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch8_eplb0_mtp3.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml | 3 ++- ...tx2dep4_gen1dep16_batch128_eplb0_mtp1.yaml | 3 ++- ...2dep4_gen1dep16_batch256_eplb256_mtp1.yaml | 3 ++- ...ctx2dep4_gen1dep8_batch512_eplb0_mtp1.yaml | 3 ++- ...ctx1dep4_gen1dep32_batch16_eplb0_mtp0.yaml | 3 ++- ...ctx1dep4_gen1dep32_batch32_eplb0_mtp0.yaml | 3 ++- ...ctx1dep4_gen1dep8_batch512_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch16_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch32_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen4tep8_batch64_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml | 3 ++- ...tx2dep4_gen1dep16_batch128_eplb0_mtp0.yaml | 3 ++- ...tx2dep4_gen1dep16_batch256_eplb0_mtp0.yaml | 3 ++- ...ctx2dep4_gen1dep32_batch64_eplb0_mtp0.yaml | 3 ++- ...x12dep4_gen1dep16_batch128_eplb0_mtp1.yaml | 3 ++- .../ctx1dep4_gen1dep32_batch2_eplb0_mtp3.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml | 3 ++- .../ctx2dep4_gen1dep32_batch4_eplb0_mtp3.yaml | 3 ++- .../ctx4dep4_gen1dep32_batch8_eplb0_mtp3.yaml | 3 ++- ...ctx6dep4_gen1dep16_batch32_eplb0_mtp3.yaml | 3 ++- ...ctx7dep4_gen1dep32_batch16_eplb0_mtp3.yaml | 3 ++- ...ctx8dep4_gen1dep16_batch64_eplb0_mtp1.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml | 3 ++- .../ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml | 3 ++- .../ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml | 3 ++- ...ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml | 3 ++- ...ctx5dep4_gen1dep16_batch64_eplb0_mtp0.yaml | 3 ++- ...ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml | 3 ++- ...tx9dep4_gen1dep16_batch128_eplb0_mtp0.yaml | 3 ++- .../disagg-B200-1p1d-dep4-dep8-c308-stp.yaml | 3 ++- .../disagg-B200-1p4d-dep4-tep8-c24-stp.yaml | 3 ++- .../disagg-B200-1p4d-dep4-tep8-c4-stp.yaml | 3 ++- .../disagg-B200-1p5d-dep4-tep4-c115-stp.yaml | 3 ++- .../disagg-B200-1p5d-dep4-tep4-c195-stp.yaml | 3 ++- .../disagg-B200-1p5d-dep4-tep4-c30-stp.yaml | 3 ++- .../disagg-B200-1p5d-dep4-tep4-c5-stp.yaml | 3 ++- .../disagg-B200-1p5d-dep4-tep4-c60-stp.yaml | 3 ++- .../disagg-B200-2p1d-dep4-dep8-c615-stp.yaml | 3 ++- .../disagg-B200-3p1d-dep4-dep8-c1127-stp.yaml | 3 ++- .../disagg-B200-4p1d-dep4-dep8-c2151-stp.yaml | 3 ++- .../8k1k/agg-gb200-low-latency-mtp2.yaml | 3 ++- .../8k1k/disagg-b200-low-latency-c1.yaml | 3 ++- .../disagg-b200-low-latency-c32-c128.yaml | 3 ++- .../8k1k/disagg-b200-low-latency-c64.yaml | 3 ++- .../8k1k/disagg-b200-low-middle-c256.yaml | 3 ++- .../8k1k/disagg-b200-low-middle-c512.yaml | 3 ++- .../8k1k/disagg-b300-high-tpt-megamoe.yaml | 3 ++- .../8k1k/disagg-b300-low-latency.yaml | 3 ++- .../8k1k/disagg-b300-mid-curve-megamoe.yaml | 3 ++- .../disagg-gb200-high-tpt-megamoe-mtp2.yaml | 3 ++- .../8k1k/disagg-gb200-high-tpt-megamoe.yaml | 3 ++- .../8k1k/disagg-gb200-low-latency-mtp2.yaml | 3 ++- .../8k1k/disagg-gb200-low-latency.yaml | 3 ++- .../8k1k/disagg-gb200-low-middle-curve.yaml | 3 ++- .../8k1k/disagg-gb200-max-tpt-megamoe.yaml | 3 ++- .../disagg-gb200-mid-curve-megamoe-mtp2.yaml | 3 ++- .../8k1k/disagg-gb200-mid-curve-megamoe.yaml | 3 ++- .../8k1k/disagg-gb300-1p6d-dep4-tp4.yaml | 3 ++- .../8k1k/disagg-gb300-1p9d-tep4-tp4.yaml | 3 ++- .../disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml | 3 ++- .../disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml | 3 ++- .../disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml | 3 ++- .../8k1k/disagg-gb300-7p2d-dep4-dep16.yaml | 3 ++- .../disagg-gb300-1p6d-dep4-tp4-agentic.yaml | 2 +- ...gb300-4p1d-dep4-dep8-24-c4096-agentic.yaml | 2 +- .../1k1k/disagg-gb200-1p1d-dep4-dep16.yaml | 3 ++- .../1k1k/disagg-gb200-1p1d-dep4-dep8.yaml | 3 ++- .../1k1k/disagg-gb200-1p4d-dep4-tp8.yaml | 3 ++- .../1k1k/disagg-gb300-1p1d-dep4-dep16.yaml | 3 ++- .../1k1k/disagg-gb300-1p1d-dep4-dep24.yaml | 3 ++- .../1k1k/disagg-gb300-1p2d-dep4-dep4.yaml | 3 ++- .../1k1k/disagg-gb300-1p7d-tep4-tp4.yaml | 3 ++- .../1k1k/disagg-gb300-2p3d-dep4-dep8.yaml | 3 ++- .../8k1k/disagg-gb200-1p4d-dep4-tep4.yaml | 3 ++- .../8k1k/disagg-gb200-1p4d-dep4-tp8.yaml | 3 ++- .../8k1k/disagg-gb200-3p1d-dep4-dep16.yaml | 3 ++- .../8k1k/disagg-gb200-6p1d-dep4-dep16.yaml | 3 ++- .../8k1k/disagg-gb200-8p1d-dep4-dep16.yaml | 3 ++- .../8k1k/disagg-gb300-1p4d-dep4-tp8.yaml | 3 ++- .../8k1k/disagg-gb300-1p8d-dep4-tp4.yaml | 3 ++- .../8k1k/disagg-gb300-2p1d-dep4-dep24.yaml | 3 ++- .../8k1k/disagg-gb300-3p1d-dep4-dep16.yaml | 3 ++- .../8k1k/disagg-gb300-4p1d-dep4-dep8.yaml | 3 ++- .../8k1k/disagg-gb300-8p1d-dep4-dep24.yaml | 3 ++- .../1k1k/1p1d-dep4-dep8-1k1k.yaml | 3 ++- .../1k1k/1p1d-dep4-tep8-1k1k.yaml | 3 ++- .../1k1k/1p2d-dep4-dep8-1k1k.yaml | 3 ++- .../1k1k/1p2d-dep4-tep8-1k1k.yaml | 3 ++- .../1k1k/1p4d-dep4-tp4-marlin-1k1k.yaml | 3 ++- .../8k1k/1p2d-dep4-dep8-8k1k.yaml | 3 ++- .../8k1k/1p2d-dep4-tep4-8k1k.yaml | 3 ++- .../8k1k/1p2d-dep4-tep8-8k1k.yaml | 3 ++- .../8k1k/2p2d-dep4-dep8-8k1k.yaml | 3 ++- .../8k1k/3p2d-dep4-dep8-8k1k.yaml | 3 ++- .../8k1k/5p2d-dep4-dep8-8k1k.yaml | 3 ++- .../1k1k/1p1d-dep2-dep4-1k1k.yaml | 3 ++- .../1k1k/1p2d-dep2-tep8-1k1k.yaml | 3 ++- .../1k1k/2p2d-dep2-tep8-1k1k.yaml | 3 ++- .../1k1k/2p3d-dep2-dep4-1k1k.yaml | 3 ++- .../1k1k/2p4d-dep2-dep4-1k1k.yaml | 3 ++- .../1k1k/4p2d-dep2-dep8-1k1k.yaml | 3 ++- .../8k1k/1p1d-dep2-dep8-8k1k.yaml | 3 ++- .../8k1k/1p1d-dep2-tep8-8k1k.yaml | 3 ++- .../8k1k/1p2d-dep2-tep8-8k1k.yaml | 3 ++- .../8k1k/2p1d-dep2-dep8-8k1k.yaml | 3 ++- .../8k1k/2p2d-dep2-tep8-8k1k.yaml | 3 ++- .../8k1k/2p4d-dep2-tep4-8k1k.yaml | 3 ++- .../8k1k/3p1d-dep2-dep8-8k1k.yaml | 3 ++- .../8k1k/3p2d-dep2-dep8-8k1k.yaml | 3 ++- .../8k1k/6p1d-dep2-dep8-8k1k.yaml | 3 ++- .../b300-fp4/1k1k/1p1d-dep2-tep8-1k1k.yaml | 3 ++- .../b300-fp4/1k1k/1p1d-dep2-tp4-1k1k.yaml | 3 ++- .../b300-fp4/1k1k/1p2d-dep2-dep4-1k1k.yaml | 3 ++- .../b300-fp4/1k1k/2p1d-dep2-dep8-1k1k.yaml | 3 ++- .../b300-fp4/1k1k/2p1d-dep2-tep8-1k1k.yaml | 3 ++- .../b300-fp4/1k1k/2p2d-dep2-tep8-1k1k.yaml | 3 ++- .../b300-fp4/1k1k/3p2d-dep2-tep8-1k1k.yaml | 3 ++- .../b300-fp4/8k1k/1p1d-tp1-tp4-8k1k.yaml | 3 ++- .../b300-fp4/8k1k/1p1d-tp1-tp8-8k1k.yaml | 3 ++- .../b300-fp4/8k1k/2p1d-tp1-tp4-8k1k.yaml | 3 ++- .../b300-fp4/8k1k/3p1d-tp1-tp4-8k1k.yaml | 3 ++- .../b300-fp4/8k1k/4p1d-tp1-tep4-8k1k.yaml | 3 ++- .../b300-fp4/8k1k/4p2d-dep2-tep4-8k1k.yaml | 3 ++- .../b300-fp4/8k1k/5p2d-tp1-dep8-8k1k.yaml | 3 ++- .../b300-fp4/8k1k/8p2d-tp1-dep8-8k1k.yaml | 3 ++- .../8k1k/mtp/1p1d-dep2-tp4-eagle3-8k1k.yaml | 3 ++- .../8k1k/mtp/1p2d-dep2-tp4-eagle3-8k1k.yaml | 3 ++- .../8k1k/mtp/1p4d-dep2-tp4-eagle3-8k1k.yaml | 3 ++- .../8k1k/mtp/1p6d-dep2-tp4-eagle3-8k1k.yaml | 3 ++- .../8k1k/mtp/2p1d-dep2-dep4-eagle3-8k1k.yaml | 3 ++- .../8k1k/mtp/2p3d-dep2-tp4-eagle3-8k1k.yaml | 3 ++- .../b300-fp8/1k1k/1p1d-dep2-tep8-1k1k.yaml | 3 ++- .../1k1k/1p1d-dep2-tp4-marlin-1k1k.yaml | 3 ++- .../b300-fp8/1k1k/1p2d-dep2-dep4-1k1k.yaml | 3 ++- .../b300-fp8/1k1k/2p1d-dep2-dep8-1k1k.yaml | 3 ++- .../b300-fp8/1k1k/2p1d-dep2-tep8-1k1k.yaml | 3 ++- .../b300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml | 3 ++- .../b300-fp8/1k1k/3p2d-dep2-tep8-1k1k.yaml | 3 ++- .../8k1k/1p1d-dep2-tp4-marlin-8k1k.yaml | 3 ++- .../b300-fp8/8k1k/1p2d-dep2-tep4-8k1k.yaml | 3 ++- .../b300-fp8/8k1k/1p4d-dep2-tep4-8k1k.yaml | 3 ++- .../b300-fp8/8k1k/1p4d-dep2-tep8-8k1k.yaml | 3 ++- .../b300-fp8/8k1k/2p2d-dep2-dep8-8k1k.yaml | 3 ++- .../b300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml | 3 ++- .../b300-fp8/8k1k/4p2d-dep2-dep8-8k1k.yaml | 3 ++- .../b300-fp8/8k1k/4p2d-dep2-tep4-8k1k.yaml | 3 ++- benchmarks/multi_node/srt_bench.sh | 6 ++--- runners/srt_slurm_benchmark_config.py | 24 ++++++++++++------- utils/bench_serving/backend_request_func.py | 4 ---- 331 files changed, 674 insertions(+), 344 deletions(-) diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx10dep4_gen1dep16_batch64_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx10dep4_gen1dep16_batch64_eplb384_mtp3.yaml index ec300feb01..9823827495 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx10dep4_gen1dep16_batch64_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx10dep4_gen1dep16_batch64_eplb384_mtp3.yaml @@ -130,7 +130,8 @@ backend: tensor_parallel_size: 16 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1229" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep8_batch512_eplb384_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep8_batch512_eplb384_mtp1.yaml index 9504350976..4ae4044a4c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep8_batch512_eplb384_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep8_batch512_eplb384_mtp1.yaml @@ -186,7 +186,8 @@ backend: tensor_parallel_size: 8 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4301" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml index 967b62e73c..b2a10d9b7c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml @@ -115,7 +115,8 @@ backend: tensor_parallel_size: 8 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "8" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml index d76b44a199..ca1b546337 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml @@ -115,7 +115,8 @@ backend: tensor_parallel_size: 4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "10" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml index 7621c206cf..c5664fbc65 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml @@ -113,7 +113,8 @@ backend: tensor_parallel_size: 4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "15" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml index 1f67c008d7..206f702c1c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml @@ -115,7 +115,8 @@ backend: tensor_parallel_size: 4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "30" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml index edd3854350..ec985ba8a3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml @@ -122,7 +122,8 @@ backend: tensor_parallel_size: 32 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "84" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml index 2bc2cbdd4f..65ea76b035 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml @@ -122,7 +122,8 @@ backend: tensor_parallel_size: 32 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "180" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml index 3beac66efc..480f920d85 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml @@ -123,7 +123,8 @@ backend: tensor_parallel_size: 32 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "333" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb384_mtp3.yaml index 50cd26ac06..506a00f1e9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb384_mtp3.yaml @@ -126,7 +126,8 @@ backend: tensor_parallel_size: 16 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "666" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep8_batch128_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep8_batch128_eplb384_mtp3.yaml index f03c2b0cf5..56c66fccca 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep8_batch128_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep8_batch128_eplb384_mtp3.yaml @@ -138,7 +138,8 @@ backend: tensor_parallel_size: 8 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1229" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep32_batch16_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep32_batch16_eplb384_mtp3.yaml index 5870e0e0ba..f9c61338f4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep32_batch16_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep32_batch16_eplb384_mtp3.yaml @@ -124,7 +124,8 @@ backend: tensor_parallel_size: 32 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "615" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx9dep4_gen1dep8_batch256_eplb384_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx9dep4_gen1dep8_batch256_eplb384_mtp1.yaml index ba65d9ee4e..bce31ae9bb 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx9dep4_gen1dep8_batch256_eplb384_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx9dep4_gen1dep8_batch256_eplb384_mtp1.yaml @@ -154,7 +154,8 @@ backend: tensor_parallel_size: 8 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "2253" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx10dep4_gen1dep8_batch512_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx10dep4_gen1dep8_batch512_eplb384_mtp0.yaml index 0db4c6c834..8e72beba3d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx10dep4_gen1dep8_batch512_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx10dep4_gen1dep8_batch512_eplb384_mtp0.yaml @@ -180,7 +180,8 @@ backend: tensor_parallel_size: 8 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4301" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml index edc9758f82..a1492263f5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml @@ -116,7 +116,8 @@ backend: tensor_parallel_size: 32 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "154" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml index a9f8989178..c39613b4a0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml @@ -109,7 +109,8 @@ backend: tensor_parallel_size: 8 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml index 1dbb0437fc..1a74b6d5bd 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml @@ -109,7 +109,8 @@ backend: tensor_parallel_size: 4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "5" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml index 1dc653de37..71c7a70a94 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml @@ -109,7 +109,8 @@ backend: tensor_parallel_size: 4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "15" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml index d0131a0224..1f86fcd29c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml @@ -109,7 +109,8 @@ backend: tensor_parallel_size: 4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "25" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml index 7eeb25c066..d1a90b9570 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml @@ -110,7 +110,8 @@ backend: tensor_parallel_size: 4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "55" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml index 993c48615c..a95b1f8276 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml @@ -117,7 +117,8 @@ backend: tensor_parallel_size: 32 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "308" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb384_mtp0.yaml index 6bc263d352..54f8d1e71d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb384_mtp0.yaml @@ -118,7 +118,8 @@ backend: tensor_parallel_size: 32 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "615" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb384_mtp0.yaml index 9303cd94a4..2b25cf30c5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb384_mtp0.yaml @@ -124,7 +124,8 @@ backend: tensor_parallel_size: 16 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1229" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb384_mtp0.yaml index a7c79ce5f4..e1ddd9ce6e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb384_mtp0.yaml @@ -120,7 +120,8 @@ backend: tensor_parallel_size: 32 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1127" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep8_batch256_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep8_batch256_eplb384_mtp0.yaml index 6f545411a0..cd16a2f8fa 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep8_batch256_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep8_batch256_eplb384_mtp0.yaml @@ -148,7 +148,8 @@ backend: tensor_parallel_size: 8 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "2253" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb384_mtp0.yaml index 7d355c69df..e9f99104eb 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb384_mtp0.yaml @@ -132,7 +132,8 @@ backend: tensor_parallel_size: 16 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "2253" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml index 55acb298f7..85e2ea7ae8 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml @@ -109,7 +109,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "8" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml index 3dcb33e574..6a10201284 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml @@ -111,7 +111,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "24" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml index 14aea33a88..795b9b3ed6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml @@ -112,7 +112,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "105" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml index 1d9b64c632..cf5a60150d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml @@ -110,7 +110,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "5" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml index eda6e401a1..db335ed3e1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml @@ -110,7 +110,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "30" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml index 26d2973090..96d7741bc5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml @@ -111,7 +111,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "60" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml index 10e9102bf2..9d85024ae8 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml @@ -111,7 +111,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "333" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml index 34742abb62..b4a7cfc197 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml @@ -118,7 +118,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1229" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml index 4b47b44101..e97190dd97 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml @@ -112,7 +112,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "666" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml index 35d8337888..2ddd6b0121 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml @@ -114,7 +114,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1229" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml index 871b7429ba..a3205e51b7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml @@ -125,7 +125,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "2253" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml index fdd8c695dc..846dfa2bfd 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml @@ -142,7 +142,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4301" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml index 9c56552fd1..ab3795dd3e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml @@ -111,7 +111,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml index 2df7b083ff..250cbb74e1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml @@ -111,7 +111,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "12" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml index 04faad8e60..5524c8792d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml @@ -111,7 +111,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "24" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml index e75f1fb65e..746e2e8a8c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml @@ -112,7 +112,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "115" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml index 0e5288717c..5f11e652c1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml @@ -110,7 +110,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "30" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml index 0a58cf8279..61f78f1210 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml @@ -111,7 +111,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "60" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml index c2875c1215..74148cdb45 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml @@ -111,7 +111,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "333" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml index ee066c1dfc..de6e745789 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml @@ -112,7 +112,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "615" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml index c28099c396..e31c124e54 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml @@ -114,7 +114,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1229" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml index 4da978b230..7c1a1d15d2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml @@ -126,7 +126,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "2253" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml index 418aed8e17..71b8046cc6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml @@ -118,7 +118,8 @@ backend: - cuda_core benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "2151" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-dep8-dep16-6-c512.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-dep8-dep16-6-c512.yaml index f46e40782e..b564127f0e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-dep8-dep16-6-c512.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-dep8-dep16-6-c512.yaml @@ -148,7 +148,8 @@ backend: benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-tp8-tp8-4-c1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-tp8-tp8-4-c1.yaml index 6305f653af..03d178f042 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-tp8-tp8-4-c1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-tp8-tp8-4-c1.yaml @@ -109,7 +109,8 @@ backend: context-length: 16384 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p2d-dep8-dep16-10-c256.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p2d-dep8-dep16-10-c256.yaml index ff0bb705a1..256c53732e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p2d-dep8-dep16-10-c256.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p2d-dep8-dep16-10-c256.yaml @@ -148,7 +148,8 @@ backend: benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p4d-dep8-tp8-10-c64.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p4d-dep8-tp8-10-c64.yaml index 84349c2775..e3bdfc6ede 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p4d-dep8-tp8-10-c64.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p4d-dep8-tp8-10-c64.yaml @@ -134,7 +134,8 @@ backend: benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "64" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-2p1d-dep8-dep16-8-c1536.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-2p1d-dep8-dep16-8-c1536.yaml index 478c91b046..7202a3e401 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-2p1d-dep8-dep16-8-c1536.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-2p1d-dep8-dep16-8-c1536.yaml @@ -148,7 +148,8 @@ backend: benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1536" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-4p1d-dep8-dep16-12-c4096.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-4p1d-dep8-dep16-12-c4096.yaml index 11434ca534..0103584eca 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-4p1d-dep8-dep16-12-c4096.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-4p1d-dep8-dep16-12-c4096.yaml @@ -148,7 +148,8 @@ backend: benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-5p1d-dep8-dep16-14-c8192.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-5p1d-dep8-dep16-14-c8192.yaml index 9962318e8a..5d3adc0bfe 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-5p1d-dep8-dep16-14-c8192.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-5p1d-dep8-dep16-14-c8192.yaml @@ -148,7 +148,8 @@ backend: benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-6p1d-dep8-dep12-15-c8192.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-6p1d-dep8-dep12-15-c8192.yaml index e88d4b7d53..939bb7bbd8 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-6p1d-dep8-dep12-15-c8192.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-6p1d-dep8-dep12-15-c8192.yaml @@ -148,7 +148,8 @@ backend: benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p1d-tp8-tp8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p1d-tp8-tp8-mtp.yaml index 0b2423a8ec..60f0235c5c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p1d-tp8-tp8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p1d-tp8-tp8-mtp.yaml @@ -113,7 +113,8 @@ backend: context-length: 16384 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p6d-dep8-tp8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p6d-dep8-tp8-mtp.yaml index 79c9a46bda..0a60722ae5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p6d-dep8-tp8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p6d-dep8-tp8-mtp.yaml @@ -120,7 +120,8 @@ backend: context-length: 16384 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-1p1d-dep8-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-1p1d-dep8-dep16-mtp.yaml index 1bf4f0e859..b76db81f99 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-1p1d-dep8-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-1p1d-dep8-dep16-mtp.yaml @@ -132,7 +132,8 @@ backend: stream-interval: 60 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-2p1d-dep8-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-2p1d-dep8-dep16-mtp.yaml index 82519e378c..1260b5e32a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-2p1d-dep8-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-2p1d-dep8-dep16-mtp.yaml @@ -132,7 +132,8 @@ backend: stream-interval: 60 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-3p1d-dep8-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-3p1d-dep8-dep16-mtp.yaml index e69c5e6043..e936f67ee1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-3p1d-dep8-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-3p1d-dep8-dep16-mtp.yaml @@ -132,7 +132,8 @@ backend: stream-interval: 60 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-4p1d-dep8-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-4p1d-dep8-dep16-mtp.yaml index 73bcecaec5..811c6a9ab6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-4p1d-dep8-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-4p1d-dep8-dep16-mtp.yaml @@ -132,7 +132,8 @@ backend: stream-interval: 60 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-5p1d-dep8-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-5p1d-dep8-dep16-mtp.yaml index 66829c4044..fe0303a8cc 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-5p1d-dep8-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-5p1d-dep8-dep16-mtp.yaml @@ -132,7 +132,8 @@ backend: stream-interval: 60 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-6p1d-dep8-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-6p1d-dep8-dep16-mtp.yaml index 34b71a9180..d1324f5758 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-6p1d-dep8-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-6p1d-dep8-dep16-mtp.yaml @@ -132,7 +132,8 @@ backend: stream-interval: 60 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-10p1d-dep4-dep32-18-c2500.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-10p1d-dep4-dep32-18-c2500.yaml index 528aa5721a..5bbfcb3076 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-10p1d-dep4-dep32-18-c2500.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-10p1d-dep4-dep32-18-c2500.yaml @@ -149,7 +149,8 @@ backend: benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "2500" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-12p1d-dep4-dep24-18-c3000.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-12p1d-dep4-dep24-18-c3000.yaml index 32a5124c2d..e4c719901f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-12p1d-dep4-dep24-18-c3000.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-12p1d-dep4-dep24-18-c3000.yaml @@ -149,7 +149,8 @@ backend: benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "3000" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-14p1d-dep4-dep16-18-c8192.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-14p1d-dep4-dep16-18-c8192.yaml index adc6b15507..469cab84b2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-14p1d-dep4-dep16-18-c8192.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-14p1d-dep4-dep16-18-c8192.yaml @@ -149,7 +149,8 @@ backend: benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-15p1d-dep4-dep12-18-c12000.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-15p1d-dep4-dep12-18-c12000.yaml index 73ad15750d..d792772ed2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-15p1d-dep4-dep12-18-c12000.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-15p1d-dep4-dep12-18-c12000.yaml @@ -149,7 +149,8 @@ backend: benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "12000" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-dep4-dep16-5-c1024.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-dep4-dep16-5-c1024.yaml index 7cdb779c71..de127e863f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-dep4-dep16-5-c1024.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-dep4-dep16-5-c1024.yaml @@ -173,7 +173,8 @@ backend: benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-tp4-tp4-2-c1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-tp4-tp4-2-c1.yaml index 9382dd6dda..0771c4b003 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-tp4-tp4-2-c1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-tp4-tp4-2-c1.yaml @@ -156,7 +156,8 @@ backend: context-length: 16384 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-8p1d-dep4-dep40-18-c2048.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-8p1d-dep4-dep40-18-c2048.yaml index 269b92e124..b63dd5fafd 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-8p1d-dep4-dep40-18-c2048.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-8p1d-dep4-dep40-18-c2048.yaml @@ -150,7 +150,8 @@ backend: benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-6p1d-dep4-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-6p1d-dep4-dep8-mtp.yaml index 39e11b7199..e7ded998f4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-6p1d-dep4-dep8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-6p1d-dep4-dep8-mtp.yaml @@ -131,7 +131,8 @@ backend: stream-interval: 60 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-8p1d-dep4-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-8p1d-dep4-dep8-mtp.yaml index a4cf477ed3..1384af6de1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-8p1d-dep4-dep8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-8p1d-dep4-dep8-mtp.yaml @@ -131,7 +131,8 @@ backend: stream-interval: 60 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p1d-tp4-tp4-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p1d-tp4-tp4-mtp.yaml index be5c4cf389..ff2a445b72 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p1d-tp4-tp4-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p1d-tp4-tp4-mtp.yaml @@ -111,7 +111,8 @@ backend: context-length: 16384 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p6d-dep4-tp4-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p6d-dep4-tp4-mtp.yaml index 5657ad84d6..e73f3ffba4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p6d-dep4-tp4-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p6d-dep4-tp4-mtp.yaml @@ -120,7 +120,8 @@ backend: context-length: 16384 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep16-mtp.yaml index f4ae3076ce..db2c0f1c83 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep16-mtp.yaml @@ -131,7 +131,8 @@ backend: stream-interval: 60 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep8-mtp.yaml index 4f9f902769..4cf1ff27c4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep8-mtp.yaml @@ -131,7 +131,8 @@ backend: stream-interval: 60 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-2p1d-dep4-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-2p1d-dep4-dep8-mtp.yaml index 17018a57ea..e937793ad1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-2p1d-dep4-dep8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-2p1d-dep4-dep8-mtp.yaml @@ -131,7 +131,8 @@ backend: stream-interval: 60 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-4p1d-dep4-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-4p1d-dep4-dep8-mtp.yaml index 15578537a4..2d652ee30a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-4p1d-dep4-dep8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-4p1d-dep4-dep8-mtp.yaml @@ -131,7 +131,8 @@ backend: stream-interval: 60 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml index 11f56972ee..bfa779d786 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml @@ -115,7 +115,8 @@ health_check: max_attempts: 360 interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml index acbec6f11a..06e658172f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml @@ -115,7 +115,8 @@ health_check: max_attempts: 360 interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_0.yaml index 5f67dfb50f..28cc45d42b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_0.yaml @@ -120,7 +120,8 @@ health_check: max_attempts: 360 interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_1.yaml index 36ceac6122..923fba978a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_1.yaml @@ -120,7 +120,8 @@ health_check: max_attempts: 360 interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_1p1d.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_1p1d.yaml index 13a1a7d9d0..f5a93f1647 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_1p1d.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_1p1d.yaml @@ -135,7 +135,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_2p1d.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_2p1d.yaml index 37225d44e8..d1504b83c1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_2p1d.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_2p1d.yaml @@ -135,7 +135,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_3p1d.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_3p1d.yaml index d8e401620a..ce7625814c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_3p1d.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_3p1d.yaml @@ -135,7 +135,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_4p1d.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_4p1d.yaml index a3923b02db..6edc9ade96 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_4p1d.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_4p1d.yaml @@ -135,7 +135,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_5p1d.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_5p1d.yaml index bcbedcb68b..afb401b71f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_5p1d.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_5p1d.yaml @@ -135,7 +135,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_4p1d_c2048.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_4p1d_c2048.yaml index fc2b0f3366..5508157f59 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_4p1d_c2048.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_4p1d_c2048.yaml @@ -111,7 +111,8 @@ health_check: max_attempts: 360 interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml index 62bbddf84d..30e7c52977 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml @@ -132,7 +132,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml index 521eddf931..d0da5d1165 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml @@ -132,7 +132,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml index 9dfc6331a8..cc5c5bf252 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml @@ -132,7 +132,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_3.yaml index ba9907840f..0e5fcb06b4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_3.yaml @@ -145,7 +145,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_4.yaml index ee2b0a2026..1da43afb6f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_4.yaml @@ -145,7 +145,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml index cd01ba6b73..6b7b60fa29 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml @@ -137,7 +137,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml index 1ec2d053a3..9c042ff16f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml @@ -137,7 +137,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_0.yaml index 6fbd1df10b..562a0d3619 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_0.yaml @@ -145,7 +145,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_1.yaml index 3c1a641a82..6cb929eb06 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_1.yaml @@ -145,7 +145,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_2.yaml index 1ee087dfa2..98864954ef 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_2.yaml @@ -145,7 +145,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_3.yaml index a03cdef22d..727071a59d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_3.yaml @@ -145,7 +145,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml index baf3aa3609..0a64b9b8c1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml @@ -137,7 +137,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml index c386534d01..208c203109 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml @@ -137,7 +137,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml index 561ac280cc..081d229cc2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml @@ -137,7 +137,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml index 76edd78e30..aee9c0abf4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml @@ -141,7 +141,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "2115" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml index 7ff2fb9e63..5a28c27612 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml @@ -141,7 +141,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "1156" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml index dde20d03d5..17a8681b4c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml @@ -141,7 +141,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "556" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml index 1a084f3876..94d3d532fe 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml @@ -136,7 +136,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "23" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml index b8cc503e54..f9864806d7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml @@ -136,7 +136,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "290" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml index ddc60e9504..850f1e03cc 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml @@ -136,7 +136,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "73" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_maxtpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_maxtpt_0.yaml index 3f7a83b6c3..ed7a8b6315 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_maxtpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_maxtpt_0.yaml @@ -141,7 +141,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "3160" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml index eb5b2467e8..7a96b283c7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml @@ -136,7 +136,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "133" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml index 07b90a7261..ae2c92d6f0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml @@ -136,7 +136,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "146" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml index f7faa73fe8..817ed54568 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml @@ -136,7 +136,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "103" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml index 242a637dfa..b59a0873a9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml @@ -136,7 +136,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "130" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml index 59a3d9fc28..5f672972b2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml @@ -136,7 +136,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "113" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_5.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_5.yaml index 595c310d30..bfa8f03048 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_5.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_5.yaml @@ -136,7 +136,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "23" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml index 8e383ee108..3f446e2651 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml @@ -141,7 +141,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "989" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml index da9815cb24..23d796f22f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml @@ -141,7 +141,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "686" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml index f5b158baee..218bf999f7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml @@ -141,7 +141,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1497" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_3.yaml index 35ae99e403..fdcc1c660d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_3.yaml @@ -141,7 +141,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "2674" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/glm5-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/glm5-mtp.yaml index b4a8864faf..b5d4880f6c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/glm5-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/glm5-mtp.yaml @@ -114,7 +114,8 @@ base: max_attempts: 360 interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf # ################# 8k1k ################# zip_override_mtp_8k1k_hightpt: diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_hightpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_hightpt_0.yaml index 854b4023a5..2d8c612e36 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_hightpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_hightpt_0.yaml @@ -124,7 +124,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml index 2ac9997129..79115655fa 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml @@ -122,7 +122,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml index 1de1730dce..3d505cdb26 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml @@ -122,7 +122,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml index 4efada1878..d70dd69dbc 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml @@ -122,7 +122,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml index 23d65f5c5e..9daf96b52b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml @@ -113,7 +113,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml index 784ced1656..d6c9d7350b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml @@ -113,7 +113,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml index f611c7ad62..2c82e02f98 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml @@ -113,7 +113,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_0.yaml index 80b88d7964..8139f6fe68 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_0.yaml @@ -124,7 +124,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_1.yaml index 734aec020f..a823495ee3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_1.yaml @@ -124,7 +124,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_2.yaml index feb1c19b73..5e0802f560 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_2.yaml @@ -124,7 +124,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_3.yaml index 29da98c953..2857c42e0b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_3.yaml @@ -124,7 +124,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml index 385063085c..7573e8c3e9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml @@ -112,7 +112,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml index 035dda552d..66784f5326 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml @@ -112,7 +112,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml index d2dadc3fa3..554fefa01d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml @@ -112,7 +112,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/glm5-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/glm5-mtp.yaml index 372c794079..b86e7385b4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/glm5-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/glm5-mtp.yaml @@ -103,7 +103,8 @@ base: max_attempts: 360 interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh req_rate: inf # MTP: EAGLE-style spec decoding is trained against chat-formatted # inputs — keep the chat template on explicitly rather than relying diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml index 31ac6edf8e..67c42f2821 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml @@ -166,7 +166,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml index 95ffd216b7..b5b935d0f3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml @@ -166,7 +166,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "64" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml index 506dede54f..f1587afbf3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml @@ -166,7 +166,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "32" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml index 43e96c0a32..c7327ee1bc 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml @@ -166,7 +166,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml index 2bcc483e9f..637a2296bf 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml @@ -166,7 +166,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "12" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-dep4-dep16.yaml index c110e15993..f298616719 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-dep4-dep16.yaml @@ -147,7 +147,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-tp4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-tp4-tp4.yaml index 143f339d0e..ac44b444b2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-tp4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-tp4-tp4.yaml @@ -113,7 +113,8 @@ backend: disaggregation-mode: "decode" benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/2p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/2p1d-dep4-dep16.yaml index 7b4ae6a03e..65f1d46489 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/2p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/2p1d-dep4-dep16.yaml @@ -147,7 +147,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/1p1d-tp4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/1p1d-tp4-tp4.yaml index d869b247a7..c858efa6d0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/1p1d-tp4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/1p1d-tp4-tp4.yaml @@ -113,7 +113,8 @@ backend: disaggregation-mode: "decode" benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/4p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/4p1d-dep4-dep16.yaml index 0eb6c58817..97ad06fbc9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/4p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/4p1d-dep4-dep16.yaml @@ -150,7 +150,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/8p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/8p1d-dep4-dep16.yaml index 58f2604a78..151aefbe60 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/8p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/8p1d-dep4-dep16.yaml @@ -153,7 +153,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml index df82d47501..4a2fdacc5d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml @@ -134,7 +134,8 @@ backend: disaggregation-mode: "decode" benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml index 544b98a962..cecf8f958f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml @@ -146,7 +146,8 @@ backend: disaggregation-transfer-backend: "mooncake" benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_0.yaml index 486efbdfe9..0b0e90ca9a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_0.yaml @@ -162,7 +162,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_1.yaml index f4a1432310..4081566648 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_1.yaml @@ -162,7 +162,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_2.yaml index bde189a965..c8dd594902 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_2.yaml @@ -162,7 +162,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_3.yaml index 0fc8db0c2b..fa6c24f27e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_3.yaml @@ -162,7 +162,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_4.yaml index 2d8bacfc92..779a72555b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_4.yaml @@ -162,7 +162,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml index 71a0ffb9f3..a6b81bc419 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml @@ -1,6 +1,6 @@ # Qwen3.5-397B-A17B-NVFP4 Disaggregated 1P1D: TP4 Prefill + TP4 Decode # Pure tensor parallel, no expert parallel (STP) -# 8k1k sa-bench concurrency sweep on GB300 +# 8k1k custom benchmark concurrency sweep on GB300 # # Values taken from ni_experiment_config of the # sa-qwen-3.5-8k1k-fp4-baseline-low-latency study, row @@ -167,7 +167,8 @@ backend: decode-log-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1x4x8x16x32x64x256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml index 00e576439a..0b5a5bb410 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml @@ -169,7 +169,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml index ecab285097..4d24e367c9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml @@ -167,7 +167,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "5120" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml index d35f444694..bf4432d114 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml @@ -167,7 +167,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "5120" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-dep4-dep16.yaml index 39e5315059..1da46c4bfc 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-dep4-dep16.yaml @@ -146,7 +146,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-tp4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-tp4-tp4.yaml index 90eaa12aae..8b5557efaf 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-tp4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-tp4-tp4.yaml @@ -113,7 +113,8 @@ backend: disaggregation-mode: "decode" benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/2p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/2p1d-dep4-dep16.yaml index 334636ebb9..c1cf903ea8 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/2p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/2p1d-dep4-dep16.yaml @@ -146,7 +146,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/1p1d-tp4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/1p1d-tp4-tp4.yaml index 0a938fb48c..15418ad74f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/1p1d-tp4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/1p1d-tp4-tp4.yaml @@ -113,7 +113,8 @@ backend: disaggregation-mode: "decode" benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/4p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/4p1d-dep4-dep16.yaml index 7ad092cfd4..a08140647a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/4p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/4p1d-dep4-dep16.yaml @@ -149,7 +149,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/8p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/8p1d-dep4-dep16.yaml index 61da0a4ad8..4a82109f9a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/8p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/8p1d-dep4-dep16.yaml @@ -152,7 +152,8 @@ backend: stream-interval: 50 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch32_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch32_eplb0_mtp3.yaml index 13da898bc4..51f71d5899 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch32_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch32_eplb0_mtp3.yaml @@ -100,7 +100,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 3 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '615' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch64_eplb0_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch64_eplb0_mtp1.yaml index 73b2d10adc..7198d70e0a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch64_eplb0_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch64_eplb0_mtp1.yaml @@ -104,7 +104,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 1 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '1229' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch16_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch16_eplb0_mtp3.yaml index 37912a792f..aa8f8944d4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch16_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch16_eplb0_mtp3.yaml @@ -98,7 +98,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 3 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '666' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch4_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch4_eplb0_mtp3.yaml index 9d09244308..6f05b2a64c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch4_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch4_eplb0_mtp3.yaml @@ -96,7 +96,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 3 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '180' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch8_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch8_eplb0_mtp3.yaml index 5af56e0452..b85e9ad6df 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch8_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch8_eplb0_mtp3.yaml @@ -97,7 +97,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 3 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '308' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch16_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch16_eplb0_mtp3.yaml index 6a40ed810c..6f453957cf 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch16_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch16_eplb0_mtp3.yaml @@ -99,7 +99,8 @@ backend: num_nextn_predict_layers: 3 allreduce_strategy: MNNVL benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '92' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml index e2919b07c9..15b16c8901 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml @@ -97,7 +97,8 @@ backend: num_nextn_predict_layers: 3 allreduce_strategy: MNNVL benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '8' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch4_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch4_eplb0_mtp3.yaml index cb043e80a7..f8b310baa3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch4_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch4_eplb0_mtp3.yaml @@ -97,7 +97,8 @@ backend: num_nextn_predict_layers: 3 allreduce_strategy: MNNVL benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '24' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch8_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch8_eplb0_mtp3.yaml index 87fc3c9899..6cd6e0613b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch8_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch8_eplb0_mtp3.yaml @@ -98,7 +98,8 @@ backend: num_nextn_predict_layers: 3 allreduce_strategy: MNNVL benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '40' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml index 3b76d83207..1214cda654 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml @@ -96,7 +96,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 3 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '10' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch128_eplb0_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch128_eplb0_mtp1.yaml index ae75aea604..af04ed713c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch128_eplb0_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch128_eplb0_mtp1.yaml @@ -112,7 +112,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 1 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '2253' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch256_eplb256_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch256_eplb256_mtp1.yaml index ed5867951f..529760fbda 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch256_eplb256_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch256_eplb256_mtp1.yaml @@ -131,7 +131,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 1 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '4301' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep8_batch512_eplb0_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep8_batch512_eplb0_mtp1.yaml index d21c50f2f9..be737f4dcb 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep8_batch512_eplb0_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep8_batch512_eplb0_mtp1.yaml @@ -160,7 +160,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 1 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '4301' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch16_eplb0_mtp0.yaml index d9b2a84884..5cee4190c5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch16_eplb0_mtp0.yaml @@ -92,7 +92,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '666' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch32_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch32_eplb0_mtp0.yaml index 9d0cfb08c3..a3d771c955 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch32_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch32_eplb0_mtp0.yaml @@ -94,7 +94,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '1229' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep8_batch512_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep8_batch512_eplb0_mtp0.yaml index f5b2378ceb..e0d4f3d838 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep8_batch512_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep8_batch512_eplb0_mtp0.yaml @@ -154,7 +154,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '4301' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch16_eplb0_mtp0.yaml index 8a4ab2820d..2b511b4109 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch16_eplb0_mtp0.yaml @@ -93,7 +93,8 @@ backend: - cuda_core allreduce_strategy: MNNVL benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '84' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml index 9a46c38543..9dae22f41b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml @@ -91,7 +91,8 @@ backend: - cuda_core allreduce_strategy: MNNVL benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '4' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch32_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch32_eplb0_mtp0.yaml index b5cf5ee77e..662e8c3d1f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch32_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch32_eplb0_mtp0.yaml @@ -95,7 +95,8 @@ backend: - cuda_core allreduce_strategy: MNNVL benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '168' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml index 61473a1f7f..6a4e756ffb 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml @@ -91,7 +91,8 @@ backend: - cuda_core allreduce_strategy: MNNVL benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '20' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch64_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch64_eplb0_mtp0.yaml index f99df622a3..e6f4e1f72b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch64_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch64_eplb0_mtp0.yaml @@ -99,7 +99,8 @@ backend: - cuda_core allreduce_strategy: MNNVL benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '284' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml index 7ef8e996f0..7ebb7b518f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml @@ -90,7 +90,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '5' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml index e3df06c70d..58fb1a4d9f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml @@ -90,7 +90,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '25' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch128_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch128_eplb0_mtp0.yaml index f3166be8c8..4ea595b27b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch128_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch128_eplb0_mtp0.yaml @@ -106,7 +106,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '2151' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch256_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch256_eplb0_mtp0.yaml index 4c588afcf6..725b2ec0a3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch256_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch256_eplb0_mtp0.yaml @@ -122,7 +122,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '4301' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep32_batch64_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep32_batch64_eplb0_mtp0.yaml index 4fa9b93291..d37017fc49 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep32_batch64_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep32_batch64_eplb0_mtp0.yaml @@ -98,7 +98,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: '2151' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep16_batch128_eplb0_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep16_batch128_eplb0_mtp1.yaml index 1d555a2504..9c5eceb2f9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep16_batch128_eplb0_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep16_batch128_eplb0_mtp1.yaml @@ -112,7 +112,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 1 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '2253' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen1dep32_batch2_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen1dep32_batch2_eplb0_mtp3.yaml index 9662511d2d..6fff3914f1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen1dep32_batch2_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen1dep32_batch2_eplb0_mtp3.yaml @@ -96,7 +96,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 3 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '90' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml index f82c0c2c6f..b946a0285c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml @@ -96,7 +96,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 3 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '5' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml index 99ff55044f..301abc5f57 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml @@ -96,7 +96,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 3 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '15' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml index d72a5d0f2e..7f63992438 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml @@ -96,7 +96,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 3 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '30' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch4_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch4_eplb0_mtp3.yaml index 5488068262..a35a90eb01 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch4_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch4_eplb0_mtp3.yaml @@ -96,7 +96,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 3 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '180' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb0_mtp3.yaml index 13905b0957..1d6a8fcc1a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb0_mtp3.yaml @@ -97,7 +97,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 3 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '333' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb0_mtp3.yaml index 0d28018a20..b2a5030c28 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb0_mtp3.yaml @@ -100,7 +100,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 3 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '615' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep32_batch16_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep32_batch16_eplb0_mtp3.yaml index 3b1c6affcd..fdcef16a4f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep32_batch16_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep32_batch16_eplb0_mtp3.yaml @@ -98,7 +98,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 3 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '666' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep16_batch64_eplb0_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep16_batch64_eplb0_mtp1.yaml index 48a2401410..3b5aac9f98 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep16_batch64_eplb0_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep16_batch64_eplb0_mtp1.yaml @@ -104,7 +104,8 @@ backend: decoding_type: MTP num_nextn_predict_layers: 1 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '1127' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml index e4ec638273..ed2fce92d1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml @@ -92,7 +92,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '105' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml index 4ed13a020f..11ce4c5556 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml @@ -90,7 +90,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '5' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml index 8c8c759822..e166e728ed 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml @@ -90,7 +90,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '10' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml index 5019bb20ed..eef78efb37 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml @@ -90,7 +90,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '25' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml index 4f3971b6c7..72dffa802b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml @@ -91,7 +91,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '50' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml index 97b70e4833..2bdd0c3670 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml @@ -91,7 +91,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '308' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml index 2a48823f9d..9831699fae 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml @@ -92,7 +92,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '615' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb0_mtp0.yaml index a2f97d489e..85a08bb3f8 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb0_mtp0.yaml @@ -98,7 +98,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '1127' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml index 461d14f5d3..b5c4f59627 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml @@ -94,7 +94,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '1229' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb0_mtp0.yaml index ad55ac6056..a8fa17a05b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb0_mtp0.yaml @@ -106,7 +106,8 @@ backend: - cutedsl - cuda_core benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: '2151' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p1d-dep4-dep8-c308-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p1d-dep4-dep8-c308-stp.yaml index 1cb6ab0de4..ad29f90c7f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p1d-dep4-dep8-c308-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p1d-dep4-dep8-c308-stp.yaml @@ -119,7 +119,8 @@ frontend: env: ETCD_LEASE_TTL: '120' benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c24-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c24-stp.yaml index 49a71471de..193efa1d3c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c24-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c24-stp.yaml @@ -115,7 +115,8 @@ frontend: env: ETCD_LEASE_TTL: '120' benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c4-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c4-stp.yaml index cabec31b18..03f58a6f2f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c4-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c4-stp.yaml @@ -115,7 +115,8 @@ frontend: env: ETCD_LEASE_TTL: '120' benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c115-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c115-stp.yaml index 298279a010..cdfcce869c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c115-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c115-stp.yaml @@ -117,7 +117,8 @@ frontend: env: ETCD_LEASE_TTL: '120' benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c195-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c195-stp.yaml index f31a7da80e..cd25750f4c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c195-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c195-stp.yaml @@ -119,7 +119,8 @@ frontend: env: ETCD_LEASE_TTL: '120' benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c30-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c30-stp.yaml index 4c4c833192..c92c05fd8b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c30-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c30-stp.yaml @@ -115,7 +115,8 @@ frontend: env: ETCD_LEASE_TTL: '120' benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c5-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c5-stp.yaml index 619c1bb709..8ac915d3c6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c5-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c5-stp.yaml @@ -115,7 +115,8 @@ frontend: env: ETCD_LEASE_TTL: '120' benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c60-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c60-stp.yaml index 1a8b4f2988..d02b72587f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c60-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c60-stp.yaml @@ -116,7 +116,8 @@ frontend: env: ETCD_LEASE_TTL: '120' benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-2p1d-dep4-dep8-c615-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-2p1d-dep4-dep8-c615-stp.yaml index abcce625e7..98d730cc08 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-2p1d-dep4-dep8-c615-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-2p1d-dep4-dep8-c615-stp.yaml @@ -123,7 +123,8 @@ frontend: env: ETCD_LEASE_TTL: '120' benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-3p1d-dep4-dep8-c1127-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-3p1d-dep4-dep8-c1127-stp.yaml index 316a88f23a..f2cd0a6ce9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-3p1d-dep4-dep8-c1127-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-3p1d-dep4-dep8-c1127-stp.yaml @@ -131,7 +131,8 @@ frontend: env: ETCD_LEASE_TTL: '120' benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-4p1d-dep4-dep8-c2151-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-4p1d-dep4-dep8-c2151-stp.yaml index 893a725cae..c94ace1e9c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-4p1d-dep4-dep8-c2151-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-4p1d-dep4-dep8-c2151-stp.yaml @@ -147,7 +147,8 @@ frontend: env: ETCD_LEASE_TTL: '120' benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/agg-gb200-low-latency-mtp2.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/agg-gb200-low-latency-mtp2.yaml index ffc8e4ea1a..1d97fcabe0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/agg-gb200-low-latency-mtp2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/agg-gb200-low-latency-mtp2.yaml @@ -68,7 +68,8 @@ backend: gpu-memory-utilization: 0.9 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c1.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c1.yaml index e81ef8631c..f57e7a6f88 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c1.yaml @@ -100,7 +100,8 @@ benchmark: isl: 8192 osl: 1024 req_rate: "inf" - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh use_chat_template: true identity: model: diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c32-c128.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c32-c128.yaml index b296440158..a76e477d0c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c32-c128.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c32-c128.yaml @@ -100,7 +100,8 @@ benchmark: isl: 8192 osl: 1024 req_rate: "inf" - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh use_chat_template: true identity: model: diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c64.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c64.yaml index b566c05fae..ce3745871b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c64.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c64.yaml @@ -100,7 +100,8 @@ benchmark: isl: 8192 osl: 1024 req_rate: "inf" - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh use_chat_template: true identity: model: diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c256.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c256.yaml index 43190836a1..da90dc1da9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c256.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c256.yaml @@ -104,7 +104,8 @@ benchmark: isl: 8192 osl: 1024 req_rate: "inf" - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh use_chat_template: true identity: model: diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c512.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c512.yaml index 283bb839f3..3ba59a7e94 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c512.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c512.yaml @@ -100,7 +100,8 @@ backend: enable-sleep-mode: true tokenizer-mode: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-high-tpt-megamoe.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-high-tpt-megamoe.yaml index 9247a31daf..b41d84056b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-high-tpt-megamoe.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-high-tpt-megamoe.yaml @@ -111,7 +111,8 @@ backend: reasoning-parser: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-low-latency.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-low-latency.yaml index 260e16fc7d..bd104db522 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-low-latency.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-low-latency.yaml @@ -108,7 +108,8 @@ backend: reasoning-parser: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1x32x64x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-mid-curve-megamoe.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-mid-curve-megamoe.yaml index 98b701790a..807f545b49 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-mid-curve-megamoe.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-mid-curve-megamoe.yaml @@ -111,7 +111,8 @@ backend: reasoning-parser: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "256x1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe-mtp2.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe-mtp2.yaml index 24b134c1b9..7116999e5b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe-mtp2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe-mtp2.yaml @@ -127,7 +127,8 @@ backend: enable-sleep-mode: true tokenizer-mode: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe.yaml index a78b83ce2e..2d32da784e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe.yaml @@ -136,7 +136,8 @@ backend: enable-sleep-mode: true tokenizer-mode: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency-mtp2.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency-mtp2.yaml index 7e0d09a0ef..52b64b1675 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency-mtp2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency-mtp2.yaml @@ -113,7 +113,8 @@ backend: enable-sleep-mode: true tokenizer-mode: deepseek_v4 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: 16x32x64 diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency.yaml index 6c5dec41d5..423a7ea9b2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency.yaml @@ -133,7 +133,8 @@ backend: enable-sleep-mode: true tokenizer-mode: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-middle-curve.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-middle-curve.yaml index 98ec9acdad..c89648d50d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-middle-curve.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-middle-curve.yaml @@ -135,7 +135,8 @@ backend: enable-sleep-mode: true tokenizer-mode: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "256x512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-max-tpt-megamoe.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-max-tpt-megamoe.yaml index db3caef440..92141b5ee1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-max-tpt-megamoe.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-max-tpt-megamoe.yaml @@ -136,7 +136,8 @@ backend: enable-sleep-mode: true tokenizer-mode: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe-mtp2.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe-mtp2.yaml index 813584eb90..521c0db9ff 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe-mtp2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe-mtp2.yaml @@ -127,7 +127,8 @@ backend: tokenizer-mode: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "128x256x512x1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe.yaml index a1f7cfd5a6..94ec4c4a5e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe.yaml @@ -136,7 +136,8 @@ backend: enable-sleep-mode: true tokenizer-mode: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "256x512x1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p6d-dep4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p6d-dep4-tp4.yaml index a541a9975f..577578c340 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p6d-dep4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p6d-dep4-tp4.yaml @@ -107,7 +107,8 @@ backend: tokenizer-mode: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p9d-tep4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p9d-tep4-tp4.yaml index e5cad82036..c04b3cf078 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p9d-tep4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p9d-tep4-tp4.yaml @@ -98,7 +98,8 @@ backend: tokenizer-mode: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "18" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml index 02adf7d4e6..0ee9a7543b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml @@ -116,7 +116,8 @@ backend: no-enable-flashinfer-autotune: true benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml index 50dbfff769..ce74045191 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml @@ -116,7 +116,8 @@ backend: no-enable-flashinfer-autotune: true benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml index 8e98947683..8b98e13bee 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml @@ -116,7 +116,8 @@ backend: no-enable-flashinfer-autotune: true benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-7p2d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-7p2d-dep4-dep16.yaml index 602275ba8f..2e118c1c94 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-7p2d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-7p2d-dep4-dep16.yaml @@ -113,7 +113,8 @@ backend: tokenizer-mode: deepseek_v4 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "3072" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb300-1p6d-dep4-tp4-agentic.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb300-1p6d-dep4-tp4-agentic.yaml index 13c8d353e3..9f95983352 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb300-1p6d-dep4-tp4-agentic.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb300-1p6d-dep4-tp4-agentic.yaml @@ -6,7 +6,7 @@ name: "svf-vllm-disagg-gb300-1p6d-dep4-tp4-agentic" # the fixed-seq-len 1p6d baseline at the same concurrency point (192). # # Divergence vs the 8k1k sibling: -# - benchmark.type: sa-bench -> custom (hands off to agentic_srt.sh) +# - benchmark.type: custom (hands off to agentic_srt.sh) # - max-model-len: removed (let vLLM derive from model config; agentic # trajectories blow past any small explicit cap) # - no-enable-prefix-caching: dropped (prefix caching MUST be on for diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb300-4p1d-dep4-dep8-24-c4096-agentic.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb300-4p1d-dep4-dep8-24-c4096-agentic.yaml index 05b779d541..4fec9c0164 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb300-4p1d-dep4-dep8-24-c4096-agentic.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb300-4p1d-dep4-dep8-24-c4096-agentic.yaml @@ -6,7 +6,7 @@ name: "svf-vllm-disagg-gb300-4p1d-dep4-dep8-24-c4096-agentic" # node. Sized for concurrency 4096 with deep_gemm_mega_moe on both workers. # # Divergence vs the 8k1k sibling: -# - benchmark.type: sa-bench -> custom (hands off to agentic_srt.sh) +# - benchmark.type: custom (hands off to agentic_srt.sh) # - max-model-len: removed (let vLLM derive from model config; agentic # trajectories blow past any small explicit cap) # - no-enable-prefix-caching: dropped (prefix caching MUST be on for diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep16.yaml index 0ee4f3e7bc..bde0e1a259 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep16.yaml @@ -93,7 +93,8 @@ backend: max-cudagraph-capture-size: 384 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "4096x6144" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep8.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep8.yaml index 40683d1b8d..c26c58be30 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep8.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep8.yaml @@ -93,7 +93,8 @@ backend: max-cudagraph-capture-size: 1024 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "4096x12288" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p4d-dep4-tp8.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p4d-dep4-tp8.yaml index bda68b9207..2b813122c9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p4d-dep4-tp8.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p4d-dep4-tp8.yaml @@ -89,7 +89,8 @@ backend: max-cudagraph-capture-size: 1024 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "4x8x32x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep16.yaml index 7188a5a082..5f5a128572 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep16.yaml @@ -92,7 +92,8 @@ backend: max-cudagraph-capture-size: 256 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "2048x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep24.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep24.yaml index 79758b546f..0ccd1766cc 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep24.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep24.yaml @@ -92,7 +92,8 @@ backend: max-cudagraph-capture-size: 128 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p2d-dep4-dep4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p2d-dep4-dep4.yaml index 15d4169abc..1d1db2fc03 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p2d-dep4-dep4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p2d-dep4-dep4.yaml @@ -92,7 +92,8 @@ backend: max-cudagraph-capture-size: 1024 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "6144" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p7d-tep4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p7d-tep4-tp4.yaml index a0ba50b871..c421bb9ea7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p7d-tep4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p7d-tep4-tp4.yaml @@ -87,7 +87,8 @@ backend: max-cudagraph-capture-size: 1024 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "8x16x32x64x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-2p3d-dep4-dep8.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-2p3d-dep4-dep8.yaml index cfa6037040..f157f8700b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-2p3d-dep4-dep8.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-2p3d-dep4-dep8.yaml @@ -92,7 +92,8 @@ backend: max-cudagraph-capture-size: 1024 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tep4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tep4.yaml index 429945c05e..3cf09eb3a8 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tep4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tep4.yaml @@ -89,7 +89,8 @@ backend: max-cudagraph-capture-size: 16 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tp8.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tp8.yaml index 55165aeb3b..e63fc93db6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tp8.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tp8.yaml @@ -89,7 +89,8 @@ backend: max-cudagraph-capture-size: 16 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4x8x16x32x256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-3p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-3p1d-dep4-dep16.yaml index b4d1e8c82e..7c8d113ee5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-3p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-3p1d-dep4-dep16.yaml @@ -92,7 +92,8 @@ backend: max-cudagraph-capture-size: 256 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-6p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-6p1d-dep4-dep16.yaml index 1f4de8df10..4d6dc7f9c5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-6p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-6p1d-dep4-dep16.yaml @@ -92,7 +92,8 @@ backend: max-cudagraph-capture-size: 512 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "3072" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-8p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-8p1d-dep4-dep16.yaml index 45a7214b2d..634b244471 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-8p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-8p1d-dep4-dep16.yaml @@ -93,7 +93,8 @@ backend: max-cudagraph-capture-size: 512 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "6144" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p4d-dep4-tp8.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p4d-dep4-tp8.yaml index 0ed5673478..6f03f2e5c8 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p4d-dep4-tp8.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p4d-dep4-tp8.yaml @@ -89,7 +89,8 @@ backend: max-cudagraph-capture-size: 1024 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p8d-dep4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p8d-dep4-tp4.yaml index 71d897b97d..fef3335660 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p8d-dep4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p8d-dep4-tp4.yaml @@ -89,7 +89,8 @@ backend: max-cudagraph-capture-size: 1024 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "32x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-2p1d-dep4-dep24.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-2p1d-dep4-dep24.yaml index b43b9f1367..04f7ee4fc3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-2p1d-dep4-dep24.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-2p1d-dep4-dep24.yaml @@ -92,7 +92,8 @@ backend: max-cudagraph-capture-size: 128 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-3p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-3p1d-dep4-dep16.yaml index 7ee94c9d10..37ce4930d0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-3p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-3p1d-dep4-dep16.yaml @@ -92,7 +92,8 @@ backend: max-cudagraph-capture-size: 128 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-4p1d-dep4-dep8.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-4p1d-dep4-dep8.yaml index 23fc7f67f0..c3caeb63a1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-4p1d-dep4-dep8.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-4p1d-dep4-dep8.yaml @@ -93,7 +93,8 @@ backend: max-cudagraph-capture-size: 512 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "3072" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-8p1d-dep4-dep24.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-8p1d-dep4-dep24.yaml index 8b2f2bf80c..ca322f7c3e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-8p1d-dep4-dep24.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-8p1d-dep4-dep24.yaml @@ -93,7 +93,8 @@ backend: max-cudagraph-capture-size: 128 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "15360" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-dep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-dep8-1k1k.yaml index 710c4d9796..7b9cd40ca4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-dep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-dep8-1k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "1024x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-tep8-1k1k.yaml index c1b9cf32d6..35d814894f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-tep8-1k1k.yaml @@ -98,7 +98,8 @@ backend: max-cudagraph-capture-size: 8196 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "128x256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-dep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-dep8-1k1k.yaml index 907633ba74..9f1e40734f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-dep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-dep8-1k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "1024x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-tep8-1k1k.yaml index 2e2bbc12a9..0deec8c06a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-tep8-1k1k.yaml @@ -98,7 +98,8 @@ backend: max-cudagraph-capture-size: 8196 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "4x16x64" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p4d-dep4-tp4-marlin-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p4d-dep4-tp4-marlin-1k1k.yaml index 429aa015ea..59ab4a338f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p4d-dep4-tp4-marlin-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p4d-dep4-tp4-marlin-1k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 8196 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "32" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-dep8-8k1k.yaml index 4129408ebd..3e6a531cd6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-dep8-8k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep4-8k1k.yaml index 5bde9ac9be..8587b7e852 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep4-8k1k.yaml @@ -98,7 +98,8 @@ backend: max-cudagraph-capture-size: 4096 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep8-8k1k.yaml index 56ed02e5e8..15e85eb784 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep8-8k1k.yaml @@ -98,7 +98,8 @@ backend: max-cudagraph-capture-size: 4096 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4x16x32x64x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/2p2d-dep4-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/2p2d-dep4-dep8-8k1k.yaml index 22ebfc7836..19f61ccef0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/2p2d-dep4-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/2p2d-dep4-dep8-8k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "512x1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/3p2d-dep4-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/3p2d-dep4-dep8-8k1k.yaml index ba1db206e4..3fa2f44ed7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/3p2d-dep4-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/3p2d-dep4-dep8-8k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/5p2d-dep4-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/5p2d-dep4-dep8-8k1k.yaml index a54ddcb226..4d1413ac2b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/5p2d-dep4-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/5p2d-dep4-dep8-8k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p1d-dep2-dep4-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p1d-dep2-dep4-1k1k.yaml index f57d7af090..a5e3886df9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p1d-dep2-dep4-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p1d-dep2-dep4-1k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p2d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p2d-dep2-tep8-1k1k.yaml index b4f457654a..18aba17e7f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p2d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p2d-dep2-tep8-1k1k.yaml @@ -98,7 +98,8 @@ backend: max-cudagraph-capture-size: 8196 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "4x16x64x128x256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml index 6bba9ea864..6b8f278c80 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml @@ -98,7 +98,8 @@ backend: max-cudagraph-capture-size: 8196 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "32" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p3d-dep2-dep4-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p3d-dep2-dep4-1k1k.yaml index de852e427a..4ca19873cd 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p3d-dep2-dep4-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p3d-dep2-dep4-1k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p4d-dep2-dep4-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p4d-dep2-dep4-1k1k.yaml index 8f7b7b140e..7449e1ea0c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p4d-dep2-dep4-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p4d-dep2-dep4-1k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/4p2d-dep2-dep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/4p2d-dep2-dep8-1k1k.yaml index f6cf6a59f0..61d575b44e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/4p2d-dep2-dep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/4p2d-dep2-dep8-1k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "1024x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-dep8-8k1k.yaml index d990d661b9..7ab2a94346 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-dep8-8k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-tep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-tep8-8k1k.yaml index d46133924c..b9da40127a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-tep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-tep8-8k1k.yaml @@ -98,7 +98,8 @@ backend: max-cudagraph-capture-size: 4096 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p2d-dep2-tep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p2d-dep2-tep8-8k1k.yaml index e8c606e271..6f0c19cdca 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p2d-dep2-tep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p2d-dep2-tep8-8k1k.yaml @@ -98,7 +98,8 @@ backend: max-cudagraph-capture-size: 4096 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "32x64x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p1d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p1d-dep2-dep8-8k1k.yaml index 02c3be14ac..f09539dcc4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p1d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p1d-dep2-dep8-8k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml index 304650d6c1..b5b40bbdca 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml @@ -98,7 +98,8 @@ backend: max-cudagraph-capture-size: 4096 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p4d-dep2-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p4d-dep2-tep4-8k1k.yaml index efea8bfacf..31b5cd3abc 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p4d-dep2-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p4d-dep2-tep4-8k1k.yaml @@ -98,7 +98,8 @@ backend: max-cudagraph-capture-size: 4096 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p1d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p1d-dep2-dep8-8k1k.yaml index 97e1ec88c8..0dbe0bf5d1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p1d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p1d-dep2-dep8-8k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p2d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p2d-dep2-dep8-8k1k.yaml index 745b2fad44..3b49f4b1d5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p2d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p2d-dep2-dep8-8k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/6p1d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/6p1d-dep2-dep8-8k1k.yaml index 9be5cc1772..1abcba608c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/6p1d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/6p1d-dep2-dep8-8k1k.yaml @@ -100,7 +100,8 @@ backend: max-cudagraph-capture-size: 2048 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tep8-1k1k.yaml index 486af05573..791fba07d8 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tep8-1k1k.yaml @@ -74,7 +74,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "4x16x64x128x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tp4-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tp4-1k1k.yaml index 532b78a103..9b568904d3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tp4-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tp4-1k1k.yaml @@ -75,7 +75,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "1x4x8x16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p2d-dep2-dep4-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p2d-dep2-dep4-1k1k.yaml index fde8442a18..65150ea97f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p2d-dep2-dep4-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p2d-dep2-dep4-1k1k.yaml @@ -76,7 +76,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-dep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-dep8-1k1k.yaml index ed3b5f9950..218da8486b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-dep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-dep8-1k1k.yaml @@ -76,7 +76,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "512x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-tep8-1k1k.yaml index 0784283b91..e1f1dc3a08 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-tep8-1k1k.yaml @@ -74,7 +74,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "32" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p2d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p2d-dep2-tep8-1k1k.yaml index 59c52da00c..12638b008e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p2d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p2d-dep2-tep8-1k1k.yaml @@ -74,7 +74,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/3p2d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/3p2d-dep2-tep8-1k1k.yaml index 7e9f7dec31..0e298026cf 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/3p2d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/3p2d-dep2-tep8-1k1k.yaml @@ -74,7 +74,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp4-8k1k.yaml index a98e9f9c83..93eb667668 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp4-8k1k.yaml @@ -66,7 +66,8 @@ health_check: max_attempts: 360 interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "8x16x24x32x48x64" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp8-8k1k.yaml index 5f699bbcf7..83117adcbf 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp8-8k1k.yaml @@ -65,7 +65,8 @@ health_check: max_attempts: 360 interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1x2x4x8x16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/2p1d-tp1-tp4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/2p1d-tp1-tp4-8k1k.yaml index 7b527ce70e..959cfadc84 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/2p1d-tp1-tp4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/2p1d-tp1-tp4-8k1k.yaml @@ -66,7 +66,8 @@ health_check: max_attempts: 360 interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/3p1d-tp1-tp4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/3p1d-tp1-tp4-8k1k.yaml index 7bf977c39e..4c9a62bbcc 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/3p1d-tp1-tp4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/3p1d-tp1-tp4-8k1k.yaml @@ -66,7 +66,8 @@ health_check: max_attempts: 360 interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p1d-tp1-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p1d-tp1-tep4-8k1k.yaml index fd11bf499f..e7d5c27a25 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p1d-tp1-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p1d-tp1-tep4-8k1k.yaml @@ -66,7 +66,8 @@ health_check: max_attempts: 360 interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p2d-dep2-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p2d-dep2-tep4-8k1k.yaml index 23c99d3282..7c9cec6fa1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p2d-dep2-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p2d-dep2-tep4-8k1k.yaml @@ -78,7 +78,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/5p2d-tp1-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/5p2d-tp1-dep8-8k1k.yaml index 55cc00b611..f7f5742a8c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/5p2d-tp1-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/5p2d-tp1-dep8-8k1k.yaml @@ -67,7 +67,8 @@ health_check: max_attempts: 360 interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/8p2d-tp1-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/8p2d-tp1-dep8-8k1k.yaml index 73525d8c42..a5d8a60755 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/8p2d-tp1-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/8p2d-tp1-dep8-8k1k.yaml @@ -67,7 +67,8 @@ health_check: max_attempts: 360 interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "768x1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p1d-dep2-tp4-eagle3-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p1d-dep2-tp4-eagle3-8k1k.yaml index 165301c53c..b6a84ee773 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p1d-dep2-tp4-eagle3-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p1d-dep2-tp4-eagle3-8k1k.yaml @@ -82,7 +82,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p2d-dep2-tp4-eagle3-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p2d-dep2-tp4-eagle3-8k1k.yaml index b32295e325..20636eaa93 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p2d-dep2-tp4-eagle3-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p2d-dep2-tp4-eagle3-8k1k.yaml @@ -82,7 +82,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p4d-dep2-tp4-eagle3-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p4d-dep2-tp4-eagle3-8k1k.yaml index ff1b0c73c7..790694a545 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p4d-dep2-tp4-eagle3-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p4d-dep2-tp4-eagle3-8k1k.yaml @@ -82,7 +82,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p6d-dep2-tp4-eagle3-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p6d-dep2-tp4-eagle3-8k1k.yaml index c5f096d64d..7973dd2bdf 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p6d-dep2-tp4-eagle3-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p6d-dep2-tp4-eagle3-8k1k.yaml @@ -82,7 +82,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p1d-dep2-dep4-eagle3-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p1d-dep2-dep4-eagle3-8k1k.yaml index 5943109612..da3ad68609 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p1d-dep2-dep4-eagle3-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p1d-dep2-dep4-eagle3-8k1k.yaml @@ -86,7 +86,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p3d-dep2-tp4-eagle3-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p3d-dep2-tp4-eagle3-8k1k.yaml index 790e6759a4..96ecdbefc2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p3d-dep2-tp4-eagle3-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p3d-dep2-tp4-eagle3-8k1k.yaml @@ -82,7 +82,8 @@ health_check: interval_seconds: 10 benchmark: - type: sa-bench + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tep8-1k1k.yaml index 5bbb133628..c9c2cf06b2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tep8-1k1k.yaml @@ -72,7 +72,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "4x16x64x128x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tp4-marlin-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tp4-marlin-1k1k.yaml index 49a60981ee..d635811972 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tp4-marlin-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tp4-marlin-1k1k.yaml @@ -74,7 +74,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "1x4x8x16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p2d-dep2-dep4-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p2d-dep2-dep4-1k1k.yaml index ef7e66d765..c8dfbecbee 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p2d-dep2-dep4-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p2d-dep2-dep4-1k1k.yaml @@ -74,7 +74,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-dep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-dep8-1k1k.yaml index 9f5aa341cf..5596247b22 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-dep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-dep8-1k1k.yaml @@ -74,7 +74,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "512x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-tep8-1k1k.yaml index 42c6e7bbc2..8882b0fc92 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-tep8-1k1k.yaml @@ -72,7 +72,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "32" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml index 3e701df05e..9b646e1d3a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml @@ -72,7 +72,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/3p2d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/3p2d-dep2-tep8-1k1k.yaml index b9a1d10588..c6bfd28bc6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/3p2d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/3p2d-dep2-tep8-1k1k.yaml @@ -72,7 +72,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 1024 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p1d-dep2-tp4-marlin-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p1d-dep2-tp4-marlin-8k1k.yaml index 04aca65862..b417471c89 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p1d-dep2-tp4-marlin-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p1d-dep2-tp4-marlin-8k1k.yaml @@ -78,7 +78,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "1x4x8x16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p2d-dep2-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p2d-dep2-tep4-8k1k.yaml index e483108986..849c81fb85 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p2d-dep2-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p2d-dep2-tep4-8k1k.yaml @@ -76,7 +76,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep4-8k1k.yaml index 30ac635a9a..736d23398f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep4-8k1k.yaml @@ -76,7 +76,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "16x32x64x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep8-8k1k.yaml index 46af72e46c..01c414f4fc 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep8-8k1k.yaml @@ -76,7 +76,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-dep8-8k1k.yaml index b1558ae34f..9326a9301f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-dep8-8k1k.yaml @@ -78,7 +78,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "256x512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml index 46aaa045da..e289ce7949 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml @@ -76,7 +76,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-dep8-8k1k.yaml index 3756103eed..49c57c199f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-dep8-8k1k.yaml @@ -78,7 +78,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-tep4-8k1k.yaml index c9f29f7852..2a7faba26d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-tep4-8k1k.yaml @@ -76,7 +76,8 @@ health_check: interval_seconds: 10 benchmark: - type: "sa-bench" + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt_bench.sh b/benchmarks/multi_node/srt_bench.sh index 63d7db3a5b..7d7466f0cc 100755 --- a/benchmarks/multi_node/srt_bench.sh +++ b/benchmarks/multi_node/srt_bench.sh @@ -1,8 +1,8 @@ #!/usr/bin/env bash # InferenceX benchmark_serving adapter for srt-slurm's custom benchmark runner. -# The recipe converter in runners/srt_slurm_benchmark_config.py supplies the -# environment below from the selected recipe's SA-Bench fields. +# The recipe hydrator in runners/srt_slurm_benchmark_config.py supplies the +# environment below from the selected recipe's throughput fields. set -euo pipefail @@ -21,7 +21,7 @@ NUM_WARMUP_MULT="${NUM_WARMUP_MULT:-2}" USE_CHAT_TEMPLATE="${USE_CHAT_TEMPLATE:-true}" DSV4="${DSV4:-false}" TRUST_REMOTE_CODE="${TRUST_REMOTE_CODE:-true}" -RESULT_DIR="/logs/sa-bench_isl_${ISL}_osl_${OSL}" +RESULT_DIR="/logs/inferencex-bench_isl_${ISL}_osl_${OSL}" ensure_bench_serving_deps() { local deps=(aiohttp numpy tqdm transformers huggingface_hub) diff --git a/runners/srt_slurm_benchmark_config.py b/runners/srt_slurm_benchmark_config.py index 0f1e6deb7c..202d65ed77 100755 --- a/runners/srt_slurm_benchmark_config.py +++ b/runners/srt_slurm_benchmark_config.py @@ -1,5 +1,5 @@ #!/usr/bin/env python3 -"""Render an srt-slurm SA-Bench recipe for InferenceX's custom benchmark.""" +"""Hydrate InferenceX srt-slurm custom benchmark recipes.""" from __future__ import annotations @@ -97,7 +97,7 @@ def stringify_concurrencies(concurrencies: list[int] | str | None) -> str: def custom_benchmark_environment(config: SrtConfig) -> dict[str, str]: - """Translate SA-Bench's typed settings to the custom wrapper contract.""" + """Translate recipe settings to the custom wrapper contract.""" benchmark = config.benchmark resources = config.resources @@ -169,10 +169,20 @@ def custom_benchmark_environment(config: SrtConfig) -> dict[str, str]: } -def convert_recipe(recipe: dict[str, Any]) -> bool: - """Convert an SA-Bench recipe in place; return whether it changed.""" +def hydrate_recipe(recipe: dict[str, Any]) -> bool: + """Hydrate an InferenceX throughput recipe; return whether it changed.""" benchmark = recipe.get("benchmark") - if not isinstance(benchmark, dict) or benchmark.get("type") != "sa-bench": + if not isinstance(benchmark, dict): + return False + + benchmark_type = benchmark.get("type") + if benchmark_type == "sa-bench": + benchmark["type"] = "custom" + benchmark["command"] = CUSTOM_BENCHMARK_COMMAND + elif not ( + benchmark_type == "custom" + and benchmark.get("command") == CUSTOM_BENCHMARK_COMMAND + ): return False cluster_config = load_cluster_config() @@ -183,8 +193,6 @@ def convert_recipe(recipe: dict[str, Any]) -> bool: environment = dict(benchmark.get("env") or {}) environment.update(custom_benchmark_environment(config)) - benchmark["type"] = "custom" - benchmark["command"] = CUSTOM_BENCHMARK_COMMAND benchmark["env"] = environment return True @@ -196,7 +204,7 @@ def render_config(config_spec: str, output_dir: Path) -> str: ) changed = False for _, recipe in variants: - changed = convert_recipe(recipe) or changed + changed = hydrate_recipe(recipe) or changed if not changed: return config_spec diff --git a/utils/bench_serving/backend_request_func.py b/utils/bench_serving/backend_request_func.py index d2e912d14c..662147ada3 100644 --- a/utils/bench_serving/backend_request_func.py +++ b/utils/bench_serving/backend_request_func.py @@ -553,10 +553,6 @@ def get_tokenizer( return MistralTokenizer.from_pretrained( str(pretrained_model_name_or_path)) else: - # DeepSeek-V4 ships tokenizer.json but its model_type is not - # registered by stock Transformers. Load that fast tokenizer directly - # so InferenceX's self-contained --dsv4 prompt encoder can be used - # without srt-slurm's separate sa_bench_tokenizers package. try: tokenizer = AutoTokenizer.from_pretrained( pretrained_model_name_or_path, From 97f9ee7f109a6064f458f5901455f9ee65bbb539 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Tue, 28 Jul 2026 11:21:36 -0500 Subject: [PATCH 4/4] refactor: make SRT benchmark entrypoint explicit MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 重命名 SRT 基准入口,并将配方参数改为显式命令参数,移除环境变量覆盖入口。 --- ...10dep4_gen1dep16_batch64_eplb384_mtp3.yaml | 2 +- ...12dep4_gen1dep8_batch512_eplb384_mtp1.yaml | 2 +- .../ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml | 2 +- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml | 2 +- .../ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml | 2 +- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml | 2 +- ...tx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml | 2 +- ...tx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml | 2 +- ...tx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml | 2 +- ...x6dep4_gen1dep16_batch32_eplb384_mtp3.yaml | 2 +- ...x7dep4_gen1dep8_batch128_eplb384_mtp3.yaml | 2 +- ...x8dep4_gen1dep32_batch16_eplb384_mtp3.yaml | 2 +- ...x9dep4_gen1dep8_batch256_eplb384_mtp1.yaml | 2 +- ...10dep4_gen1dep8_batch512_eplb384_mtp0.yaml | 2 +- ...tx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml | 2 +- .../ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml | 2 +- ...tx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml | 2 +- ...x4dep4_gen1dep32_batch16_eplb384_mtp0.yaml | 2 +- ...x5dep4_gen1dep16_batch64_eplb384_mtp0.yaml | 2 +- ...x6dep4_gen1dep32_batch32_eplb384_mtp0.yaml | 2 +- ...x6dep4_gen1dep8_batch256_eplb384_mtp0.yaml | 2 +- ...9dep4_gen1dep16_batch128_eplb384_mtp0.yaml | 2 +- .../ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml | 2 +- .../ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml | 2 +- ...ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml | 2 +- ...ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml | 2 +- ...ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml | 2 +- ...tx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml | 2 +- ...tx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml | 2 +- .../ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml | 2 +- ...ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml | 2 +- ...ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml | 2 +- ...tx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml | 2 +- ...ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml | 2 +- .../disagg-gb200-1p1d-dep8-dep16-6-c512.yaml | 2 +- .../8k1k/disagg-gb200-1p1d-tp8-tp8-4-c1.yaml | 2 +- .../disagg-gb200-1p2d-dep8-dep16-10-c256.yaml | 2 +- .../disagg-gb200-1p4d-dep8-tp8-10-c64.yaml | 2 +- .../disagg-gb200-2p1d-dep8-dep16-8-c1536.yaml | 2 +- ...disagg-gb200-4p1d-dep8-dep16-12-c4096.yaml | 2 +- ...disagg-gb200-5p1d-dep8-dep16-14-c8192.yaml | 2 +- ...disagg-gb200-6p1d-dep8-dep12-15-c8192.yaml | 2 +- ...gg-gb200-low-latency-1p1d-tp8-tp8-mtp.yaml | 2 +- ...g-gb200-low-latency-1p6d-dep8-tp8-mtp.yaml | 2 +- ...g-gb200-mid-curve-1p1d-dep8-dep16-mtp.yaml | 2 +- ...g-gb200-mid-curve-2p1d-dep8-dep16-mtp.yaml | 2 +- ...g-gb200-mid-curve-3p1d-dep8-dep16-mtp.yaml | 2 +- ...g-gb200-mid-curve-4p1d-dep8-dep16-mtp.yaml | 2 +- ...g-gb200-mid-curve-5p1d-dep8-dep16-mtp.yaml | 2 +- ...g-gb200-mid-curve-6p1d-dep8-dep16-mtp.yaml | 2 +- ...isagg-gb300-10p1d-dep4-dep32-18-c2500.yaml | 2 +- ...isagg-gb300-12p1d-dep4-dep24-18-c3000.yaml | 2 +- ...isagg-gb300-14p1d-dep4-dep16-18-c8192.yaml | 2 +- ...sagg-gb300-15p1d-dep4-dep12-18-c12000.yaml | 2 +- .../disagg-gb300-1p1d-dep4-dep16-5-c1024.yaml | 2 +- .../8k1k/disagg-gb300-1p1d-tp4-tp4-2-c1.yaml | 2 +- ...disagg-gb300-8p1d-dep4-dep40-18-c2048.yaml | 2 +- .../disagg-high-conc-6p1d-dep4-dep8-mtp.yaml | 2 +- .../disagg-high-conc-8p1d-dep4-dep8-mtp.yaml | 2 +- .../disagg-low-latency-1p1d-tp4-tp4-mtp.yaml | 2 +- .../disagg-low-latency-1p6d-dep4-tp4-mtp.yaml | 2 +- .../disagg-mid-curve-1p1d-dep4-dep16-mtp.yaml | 2 +- .../disagg-mid-curve-1p1d-dep4-dep8-mtp.yaml | 2 +- .../disagg-mid-curve-2p1d-dep4-dep8-mtp.yaml | 2 +- .../disagg-mid-curve-4p1d-dep4-dep8-mtp.yaml | 2 +- .../1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml | 2 +- .../1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml | 2 +- .../1k1k/disagg/mtp/1k1k_mtp_maxtpt_0.yaml | 2 +- .../1k1k/disagg/mtp/1k1k_mtp_maxtpt_1.yaml | 2 +- .../disagg/mtp/8k1k_mtp2_throughput_1p1d.yaml | 2 +- .../disagg/mtp/8k1k_mtp2_throughput_2p1d.yaml | 2 +- .../disagg/mtp/8k1k_mtp2_throughput_3p1d.yaml | 2 +- .../disagg/mtp/8k1k_mtp2_throughput_4p1d.yaml | 2 +- .../disagg/mtp/8k1k_mtp2_throughput_5p1d.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_4p1d_c2048.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml | 2 +- .../1k1k/disagg/mtp/1k1k_mtp_hightpt_3.yaml | 2 +- .../1k1k/disagg/mtp/1k1k_mtp_hightpt_4.yaml | 2 +- .../1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml | 2 +- .../1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_hightpt_0.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_hightpt_1.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_hightpt_2.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_hightpt_3.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_maxtpt_0.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_5.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_maxtpt_3.yaml | 2 +- .../sglang/glm5/gb200-fp4/glm5-mtp.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_hightpt_0.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml | 2 +- .../1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_hightpt_0.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_hightpt_1.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_hightpt_2.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_hightpt_3.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml | 2 +- .../sglang/glm5/gb200-fp8/glm5-mtp.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml | 2 +- .../gb200-fp8/1k1k/1p1d-dep4-dep16.yaml | 2 +- .../qwen3.5/gb200-fp8/1k1k/1p1d-tp4-tp4.yaml | 2 +- .../gb200-fp8/1k1k/2p1d-dep4-dep16.yaml | 2 +- .../qwen3.5/gb200-fp8/8k1k/1p1d-tp4-tp4.yaml | 2 +- .../gb200-fp8/8k1k/4p1d-dep4-dep16.yaml | 2 +- .../gb200-fp8/8k1k/8p1d-dep4-dep16.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_maxtpt_0.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_maxtpt_1.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_maxtpt_2.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_maxtpt_3.yaml | 2 +- .../8k1k/disagg/mtp/8k1k_mtp_maxtpt_4.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml | 2 +- .../8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml | 2 +- .../gb300-fp8/1k1k/1p1d-dep4-dep16.yaml | 2 +- .../qwen3.5/gb300-fp8/1k1k/1p1d-tp4-tp4.yaml | 2 +- .../gb300-fp8/1k1k/2p1d-dep4-dep16.yaml | 2 +- .../qwen3.5/gb300-fp8/8k1k/1p1d-tp4-tp4.yaml | 2 +- .../gb300-fp8/8k1k/4p1d-dep4-dep16.yaml | 2 +- .../gb300-fp8/8k1k/8p1d-dep4-dep16.yaml | 2 +- ...ctx1dep4_gen1dep16_batch32_eplb0_mtp3.yaml | 2 +- ...ctx1dep4_gen1dep16_batch64_eplb0_mtp1.yaml | 2 +- ...ctx1dep4_gen1dep32_batch16_eplb0_mtp3.yaml | 2 +- .../ctx1dep4_gen1dep32_batch4_eplb0_mtp3.yaml | 2 +- .../ctx1dep4_gen1dep32_batch8_eplb0_mtp3.yaml | 2 +- .../ctx1dep4_gen4tep8_batch16_eplb0_mtp3.yaml | 2 +- .../ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml | 2 +- .../ctx1dep4_gen4tep8_batch4_eplb0_mtp3.yaml | 2 +- .../ctx1dep4_gen4tep8_batch8_eplb0_mtp3.yaml | 2 +- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml | 2 +- ...tx2dep4_gen1dep16_batch128_eplb0_mtp1.yaml | 2 +- ...2dep4_gen1dep16_batch256_eplb256_mtp1.yaml | 2 +- ...ctx2dep4_gen1dep8_batch512_eplb0_mtp1.yaml | 2 +- ...ctx1dep4_gen1dep32_batch16_eplb0_mtp0.yaml | 2 +- ...ctx1dep4_gen1dep32_batch32_eplb0_mtp0.yaml | 2 +- ...ctx1dep4_gen1dep8_batch512_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen4tep8_batch16_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen4tep8_batch32_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen4tep8_batch64_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml | 2 +- ...tx2dep4_gen1dep16_batch128_eplb0_mtp0.yaml | 2 +- ...tx2dep4_gen1dep16_batch256_eplb0_mtp0.yaml | 2 +- ...ctx2dep4_gen1dep32_batch64_eplb0_mtp0.yaml | 2 +- ...x12dep4_gen1dep16_batch128_eplb0_mtp1.yaml | 2 +- .../ctx1dep4_gen1dep32_batch2_eplb0_mtp3.yaml | 2 +- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml | 2 +- .../ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml | 2 +- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml | 2 +- .../ctx2dep4_gen1dep32_batch4_eplb0_mtp3.yaml | 2 +- .../ctx4dep4_gen1dep32_batch8_eplb0_mtp3.yaml | 2 +- ...ctx6dep4_gen1dep16_batch32_eplb0_mtp3.yaml | 2 +- ...ctx7dep4_gen1dep32_batch16_eplb0_mtp3.yaml | 2 +- ...ctx8dep4_gen1dep16_batch64_eplb0_mtp1.yaml | 2 +- .../ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml | 2 +- .../ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml | 2 +- .../ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml | 2 +- ...ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml | 2 +- ...ctx5dep4_gen1dep16_batch64_eplb0_mtp0.yaml | 2 +- ...ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml | 2 +- ...tx9dep4_gen1dep16_batch128_eplb0_mtp0.yaml | 2 +- .../disagg-B200-1p1d-dep4-dep8-c308-stp.yaml | 2 +- .../disagg-B200-1p4d-dep4-tep8-c24-stp.yaml | 2 +- .../disagg-B200-1p4d-dep4-tep8-c4-stp.yaml | 2 +- .../disagg-B200-1p5d-dep4-tep4-c115-stp.yaml | 2 +- .../disagg-B200-1p5d-dep4-tep4-c195-stp.yaml | 2 +- .../disagg-B200-1p5d-dep4-tep4-c30-stp.yaml | 2 +- .../disagg-B200-1p5d-dep4-tep4-c5-stp.yaml | 2 +- .../disagg-B200-1p5d-dep4-tep4-c60-stp.yaml | 2 +- .../disagg-B200-2p1d-dep4-dep8-c615-stp.yaml | 2 +- .../disagg-B200-3p1d-dep4-dep8-c1127-stp.yaml | 2 +- .../disagg-B200-4p1d-dep4-dep8-c2151-stp.yaml | 2 +- .../8k1k/agg-gb200-low-latency-mtp2.yaml | 2 +- .../8k1k/disagg-b200-low-latency-c1.yaml | 2 +- .../disagg-b200-low-latency-c32-c128.yaml | 2 +- .../8k1k/disagg-b200-low-latency-c64.yaml | 2 +- .../8k1k/disagg-b200-low-middle-c256.yaml | 2 +- .../8k1k/disagg-b200-low-middle-c512.yaml | 2 +- .../8k1k/disagg-b300-high-tpt-megamoe.yaml | 2 +- .../8k1k/disagg-b300-low-latency.yaml | 2 +- .../8k1k/disagg-b300-mid-curve-megamoe.yaml | 2 +- .../disagg-gb200-high-tpt-megamoe-mtp2.yaml | 2 +- .../8k1k/disagg-gb200-high-tpt-megamoe.yaml | 2 +- .../8k1k/disagg-gb200-low-latency-mtp2.yaml | 2 +- .../8k1k/disagg-gb200-low-latency.yaml | 2 +- .../8k1k/disagg-gb200-low-middle-curve.yaml | 2 +- .../8k1k/disagg-gb200-max-tpt-megamoe.yaml | 2 +- .../disagg-gb200-mid-curve-megamoe-mtp2.yaml | 2 +- .../8k1k/disagg-gb200-mid-curve-megamoe.yaml | 2 +- .../8k1k/disagg-gb300-1p6d-dep4-tp4.yaml | 2 +- .../8k1k/disagg-gb300-1p9d-tep4-tp4.yaml | 2 +- .../disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml | 2 +- .../disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml | 2 +- .../disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml | 2 +- .../8k1k/disagg-gb300-7p2d-dep4-dep16.yaml | 2 +- .../1k1k/disagg-gb200-1p1d-dep4-dep16.yaml | 2 +- .../1k1k/disagg-gb200-1p1d-dep4-dep8.yaml | 2 +- .../1k1k/disagg-gb200-1p4d-dep4-tp8.yaml | 2 +- .../1k1k/disagg-gb300-1p1d-dep4-dep16.yaml | 2 +- .../1k1k/disagg-gb300-1p1d-dep4-dep24.yaml | 2 +- .../1k1k/disagg-gb300-1p2d-dep4-dep4.yaml | 2 +- .../1k1k/disagg-gb300-1p7d-tep4-tp4.yaml | 2 +- .../1k1k/disagg-gb300-2p3d-dep4-dep8.yaml | 2 +- .../8k1k/disagg-gb200-1p4d-dep4-tep4.yaml | 2 +- .../8k1k/disagg-gb200-1p4d-dep4-tp8.yaml | 2 +- .../8k1k/disagg-gb200-3p1d-dep4-dep16.yaml | 2 +- .../8k1k/disagg-gb200-6p1d-dep4-dep16.yaml | 2 +- .../8k1k/disagg-gb200-8p1d-dep4-dep16.yaml | 2 +- .../8k1k/disagg-gb300-1p4d-dep4-tp8.yaml | 2 +- .../8k1k/disagg-gb300-1p8d-dep4-tp4.yaml | 2 +- .../8k1k/disagg-gb300-2p1d-dep4-dep24.yaml | 2 +- .../8k1k/disagg-gb300-3p1d-dep4-dep16.yaml | 2 +- .../8k1k/disagg-gb300-4p1d-dep4-dep8.yaml | 2 +- .../8k1k/disagg-gb300-8p1d-dep4-dep24.yaml | 2 +- .../1k1k/1p1d-dep4-dep8-1k1k.yaml | 2 +- .../1k1k/1p1d-dep4-tep8-1k1k.yaml | 2 +- .../1k1k/1p2d-dep4-dep8-1k1k.yaml | 2 +- .../1k1k/1p2d-dep4-tep8-1k1k.yaml | 2 +- .../1k1k/1p4d-dep4-tp4-marlin-1k1k.yaml | 2 +- .../8k1k/1p2d-dep4-dep8-8k1k.yaml | 2 +- .../8k1k/1p2d-dep4-tep4-8k1k.yaml | 2 +- .../8k1k/1p2d-dep4-tep8-8k1k.yaml | 2 +- .../8k1k/2p2d-dep4-dep8-8k1k.yaml | 2 +- .../8k1k/3p2d-dep4-dep8-8k1k.yaml | 2 +- .../8k1k/5p2d-dep4-dep8-8k1k.yaml | 2 +- .../1k1k/1p1d-dep2-dep4-1k1k.yaml | 2 +- .../1k1k/1p2d-dep2-tep8-1k1k.yaml | 2 +- .../1k1k/2p2d-dep2-tep8-1k1k.yaml | 2 +- .../1k1k/2p3d-dep2-dep4-1k1k.yaml | 2 +- .../1k1k/2p4d-dep2-dep4-1k1k.yaml | 2 +- .../1k1k/4p2d-dep2-dep8-1k1k.yaml | 2 +- .../8k1k/1p1d-dep2-dep8-8k1k.yaml | 2 +- .../8k1k/1p1d-dep2-tep8-8k1k.yaml | 2 +- .../8k1k/1p2d-dep2-tep8-8k1k.yaml | 2 +- .../8k1k/2p1d-dep2-dep8-8k1k.yaml | 2 +- .../8k1k/2p2d-dep2-tep8-8k1k.yaml | 2 +- .../8k1k/2p4d-dep2-tep4-8k1k.yaml | 2 +- .../8k1k/3p1d-dep2-dep8-8k1k.yaml | 2 +- .../8k1k/3p2d-dep2-dep8-8k1k.yaml | 2 +- .../8k1k/6p1d-dep2-dep8-8k1k.yaml | 2 +- .../b300-fp4/1k1k/1p1d-dep2-tep8-1k1k.yaml | 2 +- .../b300-fp4/1k1k/1p1d-dep2-tp4-1k1k.yaml | 2 +- .../b300-fp4/1k1k/1p2d-dep2-dep4-1k1k.yaml | 2 +- .../b300-fp4/1k1k/2p1d-dep2-dep8-1k1k.yaml | 2 +- .../b300-fp4/1k1k/2p1d-dep2-tep8-1k1k.yaml | 2 +- .../b300-fp4/1k1k/2p2d-dep2-tep8-1k1k.yaml | 2 +- .../b300-fp4/1k1k/3p2d-dep2-tep8-1k1k.yaml | 2 +- .../b300-fp4/8k1k/1p1d-tp1-tp4-8k1k.yaml | 2 +- .../b300-fp4/8k1k/1p1d-tp1-tp8-8k1k.yaml | 2 +- .../b300-fp4/8k1k/2p1d-tp1-tp4-8k1k.yaml | 2 +- .../b300-fp4/8k1k/3p1d-tp1-tp4-8k1k.yaml | 2 +- .../b300-fp4/8k1k/4p1d-tp1-tep4-8k1k.yaml | 2 +- .../b300-fp4/8k1k/4p2d-dep2-tep4-8k1k.yaml | 2 +- .../b300-fp4/8k1k/5p2d-tp1-dep8-8k1k.yaml | 2 +- .../b300-fp4/8k1k/8p2d-tp1-dep8-8k1k.yaml | 2 +- .../8k1k/mtp/1p1d-dep2-tp4-eagle3-8k1k.yaml | 2 +- .../8k1k/mtp/1p2d-dep2-tp4-eagle3-8k1k.yaml | 2 +- .../8k1k/mtp/1p4d-dep2-tp4-eagle3-8k1k.yaml | 2 +- .../8k1k/mtp/1p6d-dep2-tp4-eagle3-8k1k.yaml | 2 +- .../8k1k/mtp/2p1d-dep2-dep4-eagle3-8k1k.yaml | 2 +- .../8k1k/mtp/2p3d-dep2-tp4-eagle3-8k1k.yaml | 2 +- .../b300-fp8/1k1k/1p1d-dep2-tep8-1k1k.yaml | 2 +- .../1k1k/1p1d-dep2-tp4-marlin-1k1k.yaml | 2 +- .../b300-fp8/1k1k/1p2d-dep2-dep4-1k1k.yaml | 2 +- .../b300-fp8/1k1k/2p1d-dep2-dep8-1k1k.yaml | 2 +- .../b300-fp8/1k1k/2p1d-dep2-tep8-1k1k.yaml | 2 +- .../b300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml | 2 +- .../b300-fp8/1k1k/3p2d-dep2-tep8-1k1k.yaml | 2 +- .../8k1k/1p1d-dep2-tp4-marlin-8k1k.yaml | 2 +- .../b300-fp8/8k1k/1p2d-dep2-tep4-8k1k.yaml | 2 +- .../b300-fp8/8k1k/1p4d-dep2-tep4-8k1k.yaml | 2 +- .../b300-fp8/8k1k/1p4d-dep2-tep8-8k1k.yaml | 2 +- .../b300-fp8/8k1k/2p2d-dep2-dep8-8k1k.yaml | 2 +- .../b300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml | 2 +- .../b300-fp8/8k1k/4p2d-dep2-dep8-8k1k.yaml | 2 +- .../b300-fp8/8k1k/4p2d-dep2-tep4-8k1k.yaml | 2 +- benchmarks/multi_node/srt_bench.sh | 80 ---------------- .../srt_bench_serving_entrypoint.sh | 94 +++++++++++++++++++ runners/srt_slurm_benchmark_config.py | 59 +++++++----- 329 files changed, 453 insertions(+), 432 deletions(-) delete mode 100755 benchmarks/multi_node/srt_bench.sh create mode 100755 benchmarks/multi_node/srt_bench_serving_entrypoint.sh diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx10dep4_gen1dep16_batch64_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx10dep4_gen1dep16_batch64_eplb384_mtp3.yaml index 9823827495..c8dd651d11 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx10dep4_gen1dep16_batch64_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx10dep4_gen1dep16_batch64_eplb384_mtp3.yaml @@ -131,7 +131,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1229" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep8_batch512_eplb384_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep8_batch512_eplb384_mtp1.yaml index 4ae4044a4c..afde0cab83 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep8_batch512_eplb384_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep8_batch512_eplb384_mtp1.yaml @@ -187,7 +187,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4301" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml index b2a10d9b7c..1da0c8b563 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml @@ -116,7 +116,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "8" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml index ca1b546337..15083f7578 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml @@ -116,7 +116,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "10" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml index c5664fbc65..04ae7b8563 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml @@ -114,7 +114,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "15" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml index 206f702c1c..e63fdfd01d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml @@ -116,7 +116,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "30" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml index ec985ba8a3..562b0bad61 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch2_eplb384_mtp3.yaml @@ -123,7 +123,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "84" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml index 65ea76b035..2705d4b167 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx3dep4_gen1dep32_batch4_eplb384_mtp3.yaml @@ -123,7 +123,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "180" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml index 480f920d85..6daa450c67 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb384_mtp3.yaml @@ -124,7 +124,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "333" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb384_mtp3.yaml index 506a00f1e9..b3806e169f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb384_mtp3.yaml @@ -127,7 +127,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "666" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep8_batch128_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep8_batch128_eplb384_mtp3.yaml index 56c66fccca..f6b144e299 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep8_batch128_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep8_batch128_eplb384_mtp3.yaml @@ -139,7 +139,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1229" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep32_batch16_eplb384_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep32_batch16_eplb384_mtp3.yaml index f9c61338f4..6cab42a71e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep32_batch16_eplb384_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep32_batch16_eplb384_mtp3.yaml @@ -125,7 +125,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "615" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx9dep4_gen1dep8_batch256_eplb384_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx9dep4_gen1dep8_batch256_eplb384_mtp1.yaml index bce31ae9bb..03ad62777e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx9dep4_gen1dep8_batch256_eplb384_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/MTP/ctx9dep4_gen1dep8_batch256_eplb384_mtp1.yaml @@ -155,7 +155,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "2253" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx10dep4_gen1dep8_batch512_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx10dep4_gen1dep8_batch512_eplb384_mtp0.yaml index 8e72beba3d..a53eacde9a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx10dep4_gen1dep8_batch512_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx10dep4_gen1dep8_batch512_eplb384_mtp0.yaml @@ -181,7 +181,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4301" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml index a1492263f5..8e1eb368a8 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen1dep32_batch4_eplb384_mtp0.yaml @@ -117,7 +117,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "154" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml index c39613b4a0..73f09b764b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml @@ -110,7 +110,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml index 1a74b6d5bd..943241e70f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml @@ -110,7 +110,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "5" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml index 71c7a70a94..642b6fd272 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml @@ -110,7 +110,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "15" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml index 1f86fcd29c..1f94a546ff 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml @@ -110,7 +110,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "25" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml index d1a90b9570..5f4eaffc25 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml @@ -111,7 +111,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "55" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml index a95b1f8276..7fa94e6d23 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb384_mtp0.yaml @@ -118,7 +118,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "308" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb384_mtp0.yaml index 54f8d1e71d..6b36508383 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb384_mtp0.yaml @@ -119,7 +119,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "615" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb384_mtp0.yaml index 2b25cf30c5..0961046024 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb384_mtp0.yaml @@ -125,7 +125,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1229" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb384_mtp0.yaml index e1ddd9ce6e..6493ba43aa 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb384_mtp0.yaml @@ -121,7 +121,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1127" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep8_batch256_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep8_batch256_eplb384_mtp0.yaml index cd16a2f8fa..507a169c88 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep8_batch256_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep8_batch256_eplb384_mtp0.yaml @@ -149,7 +149,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "2253" diff --git a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb384_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb384_mtp0.yaml index e9f99104eb..069a7f11da 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb384_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/DeepSeek-V4-Pro/disagg/trtllm_dynamo/gb300_mxfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb384_mtp0.yaml @@ -133,7 +133,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "2253" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml index 85e2ea7ae8..83b6633006 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml @@ -110,7 +110,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "8" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml index 6a10201284..005388ef8b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml @@ -112,7 +112,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "24" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml index 795b9b3ed6..953fb48dff 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml @@ -113,7 +113,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "105" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml index cf5a60150d..90e8e3666c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml @@ -111,7 +111,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "5" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml index db335ed3e1..81f4019f3d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml @@ -111,7 +111,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "30" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml index 96d7741bc5..71134f59c8 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml @@ -112,7 +112,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "60" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml index 9d85024ae8..9c02b8195c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml @@ -112,7 +112,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "333" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml index b4a7cfc197..b59412be8b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep16_batch64_eplb0_mtp0.yaml @@ -119,7 +119,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1229" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml index e97190dd97..0cb843d3a7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml @@ -113,7 +113,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "666" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml index 2ddd6b0121..4da77069ea 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml @@ -115,7 +115,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1229" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml index a3205e51b7..a1ca05cdf5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx7dep4_gen1dep16_batch128_eplb0_mtp0.yaml @@ -126,7 +126,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "2253" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml index 846dfa2bfd..672583855b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb200Nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch256_eplb0_mtp0.yaml @@ -143,7 +143,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4301" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml index ab3795dd3e..c11f3a36c3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml @@ -112,7 +112,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml index 250cbb74e1..3e55488a52 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch2_eplb0_mtp0.yaml @@ -112,7 +112,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "12" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml index 5524c8792d..29b843dedc 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml @@ -112,7 +112,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "24" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml index 746e2e8a8c..705f29a902 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml @@ -113,7 +113,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "115" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml index 5f11e652c1..a6957ba9ff 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml @@ -111,7 +111,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "30" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml index 61f78f1210..eff0524b1b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml @@ -112,7 +112,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "60" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml index 74148cdb45..f4a4d26f27 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml @@ -112,7 +112,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "333" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml index de6e745789..b74a10e6b2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx3dep4_gen1dep32_batch16_eplb0_mtp0.yaml @@ -113,7 +113,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "615" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml index e31c124e54..47a3fabfbf 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep32_batch32_eplb0_mtp0.yaml @@ -115,7 +115,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1229" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml index 7c1a1d15d2..bd66cfe13e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep16_batch128_eplb0_mtp0.yaml @@ -127,7 +127,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "2253" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml index 71b8046cc6..66cdd85f3d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/kimi2.5/trtllm_dynamo/disagg/gb300Nvfp4/ISL8K_OSL1K/STP/ctx8dep4_gen1dep32_batch64_eplb0_mtp0.yaml @@ -119,7 +119,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "2151" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-dep8-dep16-6-c512.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-dep8-dep16-6-c512.yaml index b564127f0e..28ec1815b2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-dep8-dep16-6-c512.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-dep8-dep16-6-c512.yaml @@ -149,7 +149,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-tp8-tp8-4-c1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-tp8-tp8-4-c1.yaml index 03d178f042..77a8fd7f41 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-tp8-tp8-4-c1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-tp8-tp8-4-c1.yaml @@ -110,7 +110,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p2d-dep8-dep16-10-c256.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p2d-dep8-dep16-10-c256.yaml index 256c53732e..e93800e443 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p2d-dep8-dep16-10-c256.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p2d-dep8-dep16-10-c256.yaml @@ -149,7 +149,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p4d-dep8-tp8-10-c64.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p4d-dep8-tp8-10-c64.yaml index e3bdfc6ede..89afabbdff 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p4d-dep8-tp8-10-c64.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p4d-dep8-tp8-10-c64.yaml @@ -135,7 +135,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "64" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-2p1d-dep8-dep16-8-c1536.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-2p1d-dep8-dep16-8-c1536.yaml index 7202a3e401..6bf374e294 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-2p1d-dep8-dep16-8-c1536.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-2p1d-dep8-dep16-8-c1536.yaml @@ -149,7 +149,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1536" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-4p1d-dep8-dep16-12-c4096.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-4p1d-dep8-dep16-12-c4096.yaml index 0103584eca..0ecb4a6d36 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-4p1d-dep8-dep16-12-c4096.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-4p1d-dep8-dep16-12-c4096.yaml @@ -149,7 +149,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-5p1d-dep8-dep16-14-c8192.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-5p1d-dep8-dep16-14-c8192.yaml index 5d3adc0bfe..3a464508af 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-5p1d-dep8-dep16-14-c8192.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-5p1d-dep8-dep16-14-c8192.yaml @@ -149,7 +149,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-6p1d-dep8-dep12-15-c8192.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-6p1d-dep8-dep12-15-c8192.yaml index 939bb7bbd8..e3ac60b586 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-6p1d-dep8-dep12-15-c8192.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-6p1d-dep8-dep12-15-c8192.yaml @@ -149,7 +149,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p1d-tp8-tp8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p1d-tp8-tp8-mtp.yaml index 60f0235c5c..e5c875cd4c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p1d-tp8-tp8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p1d-tp8-tp8-mtp.yaml @@ -114,7 +114,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p6d-dep8-tp8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p6d-dep8-tp8-mtp.yaml index 0a60722ae5..699f1cba7d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p6d-dep8-tp8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-low-latency-1p6d-dep8-tp8-mtp.yaml @@ -121,7 +121,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-1p1d-dep8-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-1p1d-dep8-dep16-mtp.yaml index b76db81f99..6edc00c0d1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-1p1d-dep8-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-1p1d-dep8-dep16-mtp.yaml @@ -133,7 +133,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-2p1d-dep8-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-2p1d-dep8-dep16-mtp.yaml index 1260b5e32a..87584f06d3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-2p1d-dep8-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-2p1d-dep8-dep16-mtp.yaml @@ -133,7 +133,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-3p1d-dep8-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-3p1d-dep8-dep16-mtp.yaml index e936f67ee1..6a7a0a9152 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-3p1d-dep8-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-3p1d-dep8-dep16-mtp.yaml @@ -133,7 +133,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-4p1d-dep8-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-4p1d-dep8-dep16-mtp.yaml index 811c6a9ab6..bb8c0ccff6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-4p1d-dep8-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-4p1d-dep8-dep16-mtp.yaml @@ -133,7 +133,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-5p1d-dep8-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-5p1d-dep8-dep16-mtp.yaml index fe0303a8cc..5a05012a25 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-5p1d-dep8-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-5p1d-dep8-dep16-mtp.yaml @@ -133,7 +133,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-6p1d-dep8-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-6p1d-dep8-dep16-mtp.yaml index d1324f5758..270d3f097f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-6p1d-dep8-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb200-mid-curve-6p1d-dep8-dep16-mtp.yaml @@ -133,7 +133,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-10p1d-dep4-dep32-18-c2500.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-10p1d-dep4-dep32-18-c2500.yaml index 5bbfcb3076..e91fd22e16 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-10p1d-dep4-dep32-18-c2500.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-10p1d-dep4-dep32-18-c2500.yaml @@ -150,7 +150,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "2500" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-12p1d-dep4-dep24-18-c3000.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-12p1d-dep4-dep24-18-c3000.yaml index e4c719901f..3635f7bacc 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-12p1d-dep4-dep24-18-c3000.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-12p1d-dep4-dep24-18-c3000.yaml @@ -150,7 +150,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "3000" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-14p1d-dep4-dep16-18-c8192.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-14p1d-dep4-dep16-18-c8192.yaml index 469cab84b2..fa8de70ae4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-14p1d-dep4-dep16-18-c8192.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-14p1d-dep4-dep16-18-c8192.yaml @@ -150,7 +150,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-15p1d-dep4-dep12-18-c12000.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-15p1d-dep4-dep12-18-c12000.yaml index d792772ed2..dfa1791e6c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-15p1d-dep4-dep12-18-c12000.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-15p1d-dep4-dep12-18-c12000.yaml @@ -150,7 +150,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "12000" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-dep4-dep16-5-c1024.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-dep4-dep16-5-c1024.yaml index de127e863f..7ea8bb37f7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-dep4-dep16-5-c1024.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-dep4-dep16-5-c1024.yaml @@ -174,7 +174,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-tp4-tp4-2-c1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-tp4-tp4-2-c1.yaml index 0771c4b003..fdb26123c5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-tp4-tp4-2-c1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-1p1d-tp4-tp4-2-c1.yaml @@ -157,7 +157,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-8p1d-dep4-dep40-18-c2048.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-8p1d-dep4-dep40-18-c2048.yaml index b63dd5fafd..6e532ee45f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-8p1d-dep4-dep40-18-c2048.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-gb300-8p1d-dep4-dep40-18-c2048.yaml @@ -151,7 +151,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-6p1d-dep4-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-6p1d-dep4-dep8-mtp.yaml index e7ded998f4..a91dd3f79e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-6p1d-dep4-dep8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-6p1d-dep4-dep8-mtp.yaml @@ -132,7 +132,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-8p1d-dep4-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-8p1d-dep4-dep8-mtp.yaml index 1384af6de1..b7a2b0e545 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-8p1d-dep4-dep8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-high-conc-8p1d-dep4-dep8-mtp.yaml @@ -132,7 +132,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p1d-tp4-tp4-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p1d-tp4-tp4-mtp.yaml index ff2a445b72..ca3f698eaa 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p1d-tp4-tp4-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p1d-tp4-tp4-mtp.yaml @@ -112,7 +112,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p6d-dep4-tp4-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p6d-dep4-tp4-mtp.yaml index e73f3ffba4..c0966bf1f5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p6d-dep4-tp4-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-low-latency-1p6d-dep4-tp4-mtp.yaml @@ -121,7 +121,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep16-mtp.yaml index db2c0f1c83..9972200ddb 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep16-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep16-mtp.yaml @@ -132,7 +132,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep8-mtp.yaml index 4cf1ff27c4..2cba35bb88 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-1p1d-dep4-dep8-mtp.yaml @@ -132,7 +132,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-2p1d-dep4-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-2p1d-dep4-dep8-mtp.yaml index e937793ad1..d2ca721142 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-2p1d-dep4-dep8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-2p1d-dep4-dep8-mtp.yaml @@ -132,7 +132,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-4p1d-dep4-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-4p1d-dep4-dep8-mtp.yaml index 2d652ee30a..6a05aca028 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-4p1d-dep4-dep8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/8k1k/disagg-mid-curve-4p1d-dep4-dep8-mtp.yaml @@ -132,7 +132,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 random_range_ratio: 0.8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml index bfa779d786..72c77ad02a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml @@ -116,7 +116,7 @@ health_check: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml index 06e658172f..ce3b1ff38c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml @@ -116,7 +116,7 @@ health_check: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_0.yaml index 28cc45d42b..841f27693c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_0.yaml @@ -121,7 +121,7 @@ health_check: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_1.yaml index 923fba978a..129d92aef5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/1k1k/disagg/mtp/1k1k_mtp_maxtpt_1.yaml @@ -121,7 +121,7 @@ health_check: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_1p1d.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_1p1d.yaml index f5a93f1647..7272666dde 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_1p1d.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_1p1d.yaml @@ -136,7 +136,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_2p1d.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_2p1d.yaml index d1504b83c1..863aa17187 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_2p1d.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_2p1d.yaml @@ -136,7 +136,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_3p1d.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_3p1d.yaml index ce7625814c..b193320ddb 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_3p1d.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_3p1d.yaml @@ -136,7 +136,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_4p1d.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_4p1d.yaml index 6edc9ade96..c762458310 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_4p1d.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_4p1d.yaml @@ -136,7 +136,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_5p1d.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_5p1d.yaml index afb401b71f..1a6282da07 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_5p1d.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp2_throughput_5p1d.yaml @@ -136,7 +136,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_4p1d_c2048.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_4p1d_c2048.yaml index 5508157f59..9532a234a6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_4p1d_c2048.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_4p1d_c2048.yaml @@ -112,7 +112,7 @@ health_check: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml index 30e7c52977..eb59efeaad 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml @@ -133,7 +133,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml index d0da5d1165..0a6414feb4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml @@ -133,7 +133,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml index cc5c5bf252..bb6f1c53ed 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/dsr1/b200-fp4/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml @@ -133,7 +133,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_3.yaml index 0e5fcb06b4..62db12c17c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_3.yaml @@ -146,7 +146,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_4.yaml index 1da43afb6f..57a6fb5e0e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_hightpt_4.yaml @@ -146,7 +146,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml index 6b7b60fa29..6357ec6378 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_0.yaml @@ -138,7 +138,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml index 9c042ff16f..8438a92675 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/1k1k/disagg/mtp/1k1k_mtp_lowlat_1.yaml @@ -138,7 +138,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_0.yaml index 562a0d3619..0e2b3700ff 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_0.yaml @@ -146,7 +146,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_1.yaml index 6cb929eb06..52cdd2943c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_1.yaml @@ -146,7 +146,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_2.yaml index 98864954ef..a0cb821f92 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_2.yaml @@ -146,7 +146,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_3.yaml index 727071a59d..fe36683375 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_hightpt_3.yaml @@ -146,7 +146,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml index 0a64b9b8c1..6678aefaa1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml @@ -138,7 +138,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml index 208c203109..d0a16ac97f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml @@ -138,7 +138,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml index 081d229cc2..c9061070ff 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5.1/gb300-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_2.yaml @@ -138,7 +138,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml index aee9c0abf4..eeafc68939 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml @@ -142,7 +142,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "2115" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml index 5a28c27612..6f9f339d12 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml @@ -142,7 +142,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "1156" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml index 17a8681b4c..1ac0e78f71 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml @@ -142,7 +142,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "556" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml index 94d3d532fe..001d4a1d4f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml @@ -137,7 +137,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "23" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml index f9864806d7..6cb8f91f58 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml @@ -137,7 +137,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "290" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml index 850f1e03cc..a0506f384b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml @@ -137,7 +137,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "73" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_maxtpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_maxtpt_0.yaml index ed7a8b6315..de6fece6e0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_maxtpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/1k1k/disagg/stp/1k1k_stp_maxtpt_0.yaml @@ -142,7 +142,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "3160" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml index 7a96b283c7..00bd6d8a60 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml @@ -137,7 +137,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "133" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml index ae2c92d6f0..e1a3395fc7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml @@ -137,7 +137,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "146" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml index 817ed54568..5dde716dc3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml @@ -137,7 +137,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "103" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml index b59a0873a9..d6247f91b2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml @@ -137,7 +137,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "130" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml index 5f672972b2..48ceba5c8c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml @@ -137,7 +137,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "113" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_5.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_5.yaml index bfa8f03048..4d2bb3393c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_5.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_5.yaml @@ -137,7 +137,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "23" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml index 3f446e2651..e0a107c957 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml @@ -142,7 +142,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "989" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml index 23d796f22f..27b7bfc31f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml @@ -142,7 +142,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "686" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml index 218bf999f7..9cc26f4c79 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml @@ -142,7 +142,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1497" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_3.yaml index fdcc1c660d..33de27f3b0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_3.yaml @@ -142,7 +142,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "2674" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/glm5-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/glm5-mtp.yaml index b5d4880f6c..7a8cc0a7aa 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/glm5-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp4/glm5-mtp.yaml @@ -115,7 +115,7 @@ base: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf # ################# 8k1k ################# zip_override_mtp_8k1k_hightpt: diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_hightpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_hightpt_0.yaml index 2d8c612e36..a663b68066 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_hightpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_hightpt_0.yaml @@ -125,7 +125,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml index 79115655fa..8bb57af814 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_0.yaml @@ -123,7 +123,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml index 3d505cdb26..44e86dabb8 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_1.yaml @@ -123,7 +123,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml index d70dd69dbc..32506810e3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_2.yaml @@ -123,7 +123,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml index 9daf96b52b..70302685f7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_3.yaml @@ -114,7 +114,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml index d6c9d7350b..e1540ffaad 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_4.yaml @@ -114,7 +114,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml index 2c82e02f98..acf7121b35 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/1k1k/disagg/stp/1k1k_stp_lowlat_5.yaml @@ -114,7 +114,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 1024 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_0.yaml index 8139f6fe68..3ef558666c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_0.yaml @@ -125,7 +125,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_1.yaml index a823495ee3..04072d4bdf 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_1.yaml @@ -125,7 +125,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_2.yaml index 5e0802f560..f715e91a17 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_2.yaml @@ -125,7 +125,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_3.yaml index 2857c42e0b..a805e75acb 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_hightpt_3.yaml @@ -125,7 +125,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml index 7573e8c3e9..a9bbaa821b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml @@ -113,7 +113,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml index 66784f5326..ad7b1b07a9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml @@ -113,7 +113,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml index 554fefa01d..d27af50f81 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml @@ -113,7 +113,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf isl: 8192 osl: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/glm5-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/glm5-mtp.yaml index b86e7385b4..b62dd3c51d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/glm5-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb200-fp8/glm5-mtp.yaml @@ -104,7 +104,7 @@ base: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh req_rate: inf # MTP: EAGLE-style spec decoding is trained against chat-formatted # inputs — keep the chat template on explicitly rather than relying diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml index 67c42f2821..18517d3a04 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml @@ -167,7 +167,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml index b5b935d0f3..ca8a45f14f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_1.yaml @@ -167,7 +167,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "64" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml index f1587afbf3..26d4bf46af 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_2.yaml @@ -167,7 +167,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "32" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml index c7327ee1bc..40f529081f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_3.yaml @@ -167,7 +167,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml index 637a2296bf..216c140896 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/glm5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_4.yaml @@ -167,7 +167,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "12" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-dep4-dep16.yaml index f298616719..0fd906e6a6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-dep4-dep16.yaml @@ -148,7 +148,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-tp4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-tp4-tp4.yaml index ac44b444b2..fbdd081d35 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-tp4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/1p1d-tp4-tp4.yaml @@ -114,7 +114,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/2p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/2p1d-dep4-dep16.yaml index 65f1d46489..85f8a32b1b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/2p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/1k1k/2p1d-dep4-dep16.yaml @@ -148,7 +148,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/1p1d-tp4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/1p1d-tp4-tp4.yaml index c858efa6d0..42919159e0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/1p1d-tp4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/1p1d-tp4-tp4.yaml @@ -114,7 +114,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/4p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/4p1d-dep4-dep16.yaml index 97ad06fbc9..d7ecf653fd 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/4p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/4p1d-dep4-dep16.yaml @@ -151,7 +151,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/8p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/8p1d-dep4-dep16.yaml index 151aefbe60..54704714d2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/8p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/8p1d-dep4-dep16.yaml @@ -154,7 +154,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml index 4a2fdacc5d..f632db58a0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_0.yaml @@ -135,7 +135,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml index cecf8f958f..6547ba7caf 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_lowlat_1.yaml @@ -147,7 +147,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_0.yaml index 0b0e90ca9a..67923f85fe 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_0.yaml @@ -163,7 +163,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_1.yaml index 4081566648..cf5837a22a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_1.yaml @@ -163,7 +163,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_2.yaml index c8dd594902..3f4e37e645 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_2.yaml @@ -163,7 +163,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_3.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_3.yaml index fa6c24f27e..15c1085447 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_3.yaml @@ -163,7 +163,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_4.yaml index 779a72555b..75f001c463 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb200-fp8/8k1k/disagg/mtp/8k1k_mtp_maxtpt_4.yaml @@ -163,7 +163,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml index a6b81bc419..55c7b4f0cb 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_lowlat_0.yaml @@ -168,7 +168,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1x4x8x16x32x64x256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml index 0b5a5bb410..a529d0e6b7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_0.yaml @@ -170,7 +170,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml index 4d24e367c9..1b7428d593 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_1.yaml @@ -168,7 +168,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "5120" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml index bf4432d114..8cfe8ccff6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp4/8k1k/disagg/stp/8k1k_stp_maxtpt_2.yaml @@ -168,7 +168,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "5120" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-dep4-dep16.yaml index 1da46c4bfc..f1c82b84fd 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-dep4-dep16.yaml @@ -147,7 +147,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-tp4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-tp4-tp4.yaml index 8b5557efaf..ba5fdd2049 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-tp4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/1p1d-tp4-tp4.yaml @@ -114,7 +114,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/2p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/2p1d-dep4-dep16.yaml index c1cf903ea8..abca86c4d4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/2p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/1k1k/2p1d-dep4-dep16.yaml @@ -147,7 +147,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/1p1d-tp4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/1p1d-tp4-tp4.yaml index 15418ad74f..7775da8e94 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/1p1d-tp4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/1p1d-tp4-tp4.yaml @@ -114,7 +114,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/4p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/4p1d-dep4-dep16.yaml index a08140647a..0f841bf0c3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/4p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/4p1d-dep4-dep16.yaml @@ -150,7 +150,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/8p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/8p1d-dep4-dep16.yaml index 4a82109f9a..cae55f0a0e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/8p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/sglang/qwen3.5/gb300-fp8/8k1k/8p1d-dep4-dep16.yaml @@ -153,7 +153,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch32_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch32_eplb0_mtp3.yaml index 51f71d5899..bab412692b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch32_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch32_eplb0_mtp3.yaml @@ -101,7 +101,7 @@ backend: num_nextn_predict_layers: 3 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '615' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch64_eplb0_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch64_eplb0_mtp1.yaml index 7198d70e0a..e34c1947fd 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch64_eplb0_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep16_batch64_eplb0_mtp1.yaml @@ -105,7 +105,7 @@ backend: num_nextn_predict_layers: 1 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '1229' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch16_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch16_eplb0_mtp3.yaml index aa8f8944d4..6bbc79017b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch16_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch16_eplb0_mtp3.yaml @@ -99,7 +99,7 @@ backend: num_nextn_predict_layers: 3 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '666' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch4_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch4_eplb0_mtp3.yaml index 6f05b2a64c..316dfa6e85 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch4_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch4_eplb0_mtp3.yaml @@ -97,7 +97,7 @@ backend: num_nextn_predict_layers: 3 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '180' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch8_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch8_eplb0_mtp3.yaml index b85e9ad6df..cfc206093a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch8_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen1dep32_batch8_eplb0_mtp3.yaml @@ -98,7 +98,7 @@ backend: num_nextn_predict_layers: 3 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '308' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch16_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch16_eplb0_mtp3.yaml index 6f453957cf..8f93b714ed 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch16_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch16_eplb0_mtp3.yaml @@ -100,7 +100,7 @@ backend: allreduce_strategy: MNNVL benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '92' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml index 15b16c8901..d24277a4ac 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch1_eplb0_mtp3.yaml @@ -98,7 +98,7 @@ backend: allreduce_strategy: MNNVL benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '8' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch4_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch4_eplb0_mtp3.yaml index f8b310baa3..0e96e939c6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch4_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch4_eplb0_mtp3.yaml @@ -98,7 +98,7 @@ backend: allreduce_strategy: MNNVL benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '24' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch8_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch8_eplb0_mtp3.yaml index 6cd6e0613b..e215a5d18e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch8_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen4tep8_batch8_eplb0_mtp3.yaml @@ -99,7 +99,7 @@ backend: allreduce_strategy: MNNVL benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '40' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml index 1214cda654..1f03ab345e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml @@ -97,7 +97,7 @@ backend: num_nextn_predict_layers: 3 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '10' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch128_eplb0_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch128_eplb0_mtp1.yaml index af04ed713c..ff7eb7316a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch128_eplb0_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch128_eplb0_mtp1.yaml @@ -113,7 +113,7 @@ backend: num_nextn_predict_layers: 1 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '2253' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch256_eplb256_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch256_eplb256_mtp1.yaml index 529760fbda..529ccd19a6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch256_eplb256_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep16_batch256_eplb256_mtp1.yaml @@ -132,7 +132,7 @@ backend: num_nextn_predict_layers: 1 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '4301' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep8_batch512_eplb0_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep8_batch512_eplb0_mtp1.yaml index be737f4dcb..4a11364b69 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep8_batch512_eplb0_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/MTP/ctx2dep4_gen1dep8_batch512_eplb0_mtp1.yaml @@ -161,7 +161,7 @@ backend: num_nextn_predict_layers: 1 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '4301' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch16_eplb0_mtp0.yaml index 5cee4190c5..da3839cf1b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch16_eplb0_mtp0.yaml @@ -93,7 +93,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '666' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch32_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch32_eplb0_mtp0.yaml index a3d771c955..18107fa606 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch32_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep32_batch32_eplb0_mtp0.yaml @@ -95,7 +95,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '1229' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep8_batch512_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep8_batch512_eplb0_mtp0.yaml index e0d4f3d838..9be5d9c83a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep8_batch512_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen1dep8_batch512_eplb0_mtp0.yaml @@ -155,7 +155,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '4301' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch16_eplb0_mtp0.yaml index 2b511b4109..9b256d1b8b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch16_eplb0_mtp0.yaml @@ -94,7 +94,7 @@ backend: allreduce_strategy: MNNVL benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '84' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml index 9dae22f41b..c092e01b87 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch1_eplb0_mtp0.yaml @@ -92,7 +92,7 @@ backend: allreduce_strategy: MNNVL benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '4' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch32_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch32_eplb0_mtp0.yaml index 662e8c3d1f..67e0f34879 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch32_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch32_eplb0_mtp0.yaml @@ -96,7 +96,7 @@ backend: allreduce_strategy: MNNVL benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '168' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml index 6a4e756ffb..0913d27ded 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch4_eplb0_mtp0.yaml @@ -92,7 +92,7 @@ backend: allreduce_strategy: MNNVL benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '20' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch64_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch64_eplb0_mtp0.yaml index e6f4e1f72b..d87df1fdec 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch64_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen4tep8_batch64_eplb0_mtp0.yaml @@ -100,7 +100,7 @@ backend: allreduce_strategy: MNNVL benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '284' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml index 7ebb7b518f..0276e0f9ec 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml @@ -91,7 +91,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '5' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml index 58fb1a4d9f..6c33d6aba0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml @@ -91,7 +91,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '25' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch128_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch128_eplb0_mtp0.yaml index 4ea595b27b..c9eaf58bf7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch128_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch128_eplb0_mtp0.yaml @@ -107,7 +107,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '2151' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch256_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch256_eplb0_mtp0.yaml index 725b2ec0a3..1446d3dadd 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch256_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep16_batch256_eplb0_mtp0.yaml @@ -123,7 +123,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '4301' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep32_batch64_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep32_batch64_eplb0_mtp0.yaml index d37017fc49..ed6bf8693d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep32_batch64_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL1K_OSL1K/STP/ctx2dep4_gen1dep32_batch64_eplb0_mtp0.yaml @@ -99,7 +99,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: '2151' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep16_batch128_eplb0_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep16_batch128_eplb0_mtp1.yaml index 9c5eceb2f9..3d4e5229a6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep16_batch128_eplb0_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx12dep4_gen1dep16_batch128_eplb0_mtp1.yaml @@ -113,7 +113,7 @@ backend: num_nextn_predict_layers: 1 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '2253' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen1dep32_batch2_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen1dep32_batch2_eplb0_mtp3.yaml index 6fff3914f1..e409cd1d1d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen1dep32_batch2_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen1dep32_batch2_eplb0_mtp3.yaml @@ -97,7 +97,7 @@ backend: num_nextn_predict_layers: 3 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '90' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml index b946a0285c..637690d5b3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch1_eplb0_mtp3.yaml @@ -97,7 +97,7 @@ backend: num_nextn_predict_layers: 3 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '5' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml index 301abc5f57..980fc2206d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch2_eplb0_mtp3.yaml @@ -97,7 +97,7 @@ backend: num_nextn_predict_layers: 3 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '15' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml index 7f63992438..0e4b95f827 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx1dep4_gen5tep4_batch4_eplb0_mtp3.yaml @@ -97,7 +97,7 @@ backend: num_nextn_predict_layers: 3 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '30' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch4_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch4_eplb0_mtp3.yaml index a35a90eb01..36a9fc2683 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch4_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx2dep4_gen1dep32_batch4_eplb0_mtp3.yaml @@ -97,7 +97,7 @@ backend: num_nextn_predict_layers: 3 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '180' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb0_mtp3.yaml index 1d6a8fcc1a..5a481e5d78 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx4dep4_gen1dep32_batch8_eplb0_mtp3.yaml @@ -98,7 +98,7 @@ backend: num_nextn_predict_layers: 3 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '333' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb0_mtp3.yaml index b2a5030c28..67b69cc205 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx6dep4_gen1dep16_batch32_eplb0_mtp3.yaml @@ -101,7 +101,7 @@ backend: num_nextn_predict_layers: 3 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '615' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep32_batch16_eplb0_mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep32_batch16_eplb0_mtp3.yaml index fdcef16a4f..144a0ef948 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep32_batch16_eplb0_mtp3.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx7dep4_gen1dep32_batch16_eplb0_mtp3.yaml @@ -99,7 +99,7 @@ backend: num_nextn_predict_layers: 3 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '666' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep16_batch64_eplb0_mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep16_batch64_eplb0_mtp1.yaml index 3b5aac9f98..3ca18f15d6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep16_batch64_eplb0_mtp1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/MTP/ctx8dep4_gen1dep16_batch64_eplb0_mtp1.yaml @@ -105,7 +105,7 @@ backend: num_nextn_predict_layers: 1 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '1127' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml index ed2fce92d1..0a5f8ae589 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch16_eplb0_mtp0.yaml @@ -93,7 +93,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '105' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml index 11ce4c5556..f53536c980 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch1_eplb0_mtp0.yaml @@ -91,7 +91,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '5' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml index e166e728ed..e2c826b125 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch2_eplb0_mtp0.yaml @@ -91,7 +91,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '10' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml index eef78efb37..723876fc52 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch4_eplb0_mtp0.yaml @@ -91,7 +91,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '25' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml index 72dffa802b..1527c57f91 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx1dep4_gen5tep4_batch8_eplb0_mtp0.yaml @@ -92,7 +92,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '50' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml index 2bdd0c3670..6ca30a752a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx2dep4_gen1dep32_batch8_eplb0_mtp0.yaml @@ -92,7 +92,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '308' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml index 9831699fae..0580e5ddd7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx4dep4_gen1dep32_batch16_eplb0_mtp0.yaml @@ -93,7 +93,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '615' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb0_mtp0.yaml index 85a08bb3f8..291f7e00c7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx5dep4_gen1dep16_batch64_eplb0_mtp0.yaml @@ -99,7 +99,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '1127' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml index b5c4f59627..77affe86d6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx6dep4_gen1dep32_batch32_eplb0_mtp0.yaml @@ -95,7 +95,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '1229' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb0_mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb0_mtp0.yaml index a8fa17a05b..0dfe7f1190 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb0_mtp0.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/glm5/disagg/trtllm_dynamo/gb200_nvfp4/ISL8K_OSL1K/STP/ctx9dep4_gen1dep16_batch128_eplb0_mtp0.yaml @@ -107,7 +107,7 @@ backend: - cuda_core benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: '2151' diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p1d-dep4-dep8-c308-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p1d-dep4-dep8-c308-stp.yaml index ad29f90c7f..f22f7a284e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p1d-dep4-dep8-c308-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p1d-dep4-dep8-c308-stp.yaml @@ -120,7 +120,7 @@ frontend: ETCD_LEASE_TTL: '120' benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c24-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c24-stp.yaml index 193efa1d3c..595eefea58 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c24-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c24-stp.yaml @@ -116,7 +116,7 @@ frontend: ETCD_LEASE_TTL: '120' benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c4-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c4-stp.yaml index 03f58a6f2f..a9b6614dc8 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c4-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p4d-dep4-tep8-c4-stp.yaml @@ -116,7 +116,7 @@ frontend: ETCD_LEASE_TTL: '120' benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c115-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c115-stp.yaml index cdfcce869c..f548fb4139 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c115-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c115-stp.yaml @@ -118,7 +118,7 @@ frontend: ETCD_LEASE_TTL: '120' benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c195-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c195-stp.yaml index cd25750f4c..c21aa53211 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c195-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c195-stp.yaml @@ -120,7 +120,7 @@ frontend: ETCD_LEASE_TTL: '120' benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c30-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c30-stp.yaml index c92c05fd8b..a802a8c9b4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c30-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c30-stp.yaml @@ -116,7 +116,7 @@ frontend: ETCD_LEASE_TTL: '120' benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c5-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c5-stp.yaml index 8ac915d3c6..18ec20c235 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c5-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c5-stp.yaml @@ -116,7 +116,7 @@ frontend: ETCD_LEASE_TTL: '120' benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c60-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c60-stp.yaml index d02b72587f..3d2a791556 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c60-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-1p5d-dep4-tep4-c60-stp.yaml @@ -117,7 +117,7 @@ frontend: ETCD_LEASE_TTL: '120' benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-2p1d-dep4-dep8-c615-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-2p1d-dep4-dep8-c615-stp.yaml index 98d730cc08..3acf877ea4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-2p1d-dep4-dep8-c615-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-2p1d-dep4-dep8-c615-stp.yaml @@ -124,7 +124,7 @@ frontend: ETCD_LEASE_TTL: '120' benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-3p1d-dep4-dep8-c1127-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-3p1d-dep4-dep8-c1127-stp.yaml index f2cd0a6ce9..50be2ab1e3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-3p1d-dep4-dep8-c1127-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-3p1d-dep4-dep8-c1127-stp.yaml @@ -132,7 +132,7 @@ frontend: ETCD_LEASE_TTL: '120' benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-4p1d-dep4-dep8-c2151-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-4p1d-dep4-dep8-c2151-stp.yaml index c94ace1e9c..5999e094cb 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-4p1d-dep4-dep8-c2151-stp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/trtllm/kimi-k2.5/disagg/trtllm_dynamo/b200-fp4/disagg-B200-4p1d-dep4-dep8-c2151-stp.yaml @@ -148,7 +148,7 @@ frontend: ETCD_LEASE_TTL: '120' benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/agg-gb200-low-latency-mtp2.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/agg-gb200-low-latency-mtp2.yaml index 1d97fcabe0..388a23f772 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/agg-gb200-low-latency-mtp2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/agg-gb200-low-latency-mtp2.yaml @@ -69,7 +69,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c1.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c1.yaml index f57e7a6f88..277452978a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c1.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c1.yaml @@ -101,7 +101,7 @@ benchmark: osl: 1024 req_rate: "inf" type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh use_chat_template: true identity: model: diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c32-c128.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c32-c128.yaml index a76e477d0c..368a0c2fa2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c32-c128.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c32-c128.yaml @@ -101,7 +101,7 @@ benchmark: osl: 1024 req_rate: "inf" type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh use_chat_template: true identity: model: diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c64.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c64.yaml index ce3745871b..03b72308d3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c64.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-latency-c64.yaml @@ -101,7 +101,7 @@ benchmark: osl: 1024 req_rate: "inf" type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh use_chat_template: true identity: model: diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c256.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c256.yaml index da90dc1da9..74574f3e89 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c256.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c256.yaml @@ -105,7 +105,7 @@ benchmark: osl: 1024 req_rate: "inf" type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh use_chat_template: true identity: model: diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c512.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c512.yaml index 3ba59a7e94..89fa3b12dc 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c512.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b200-low-middle-c512.yaml @@ -101,7 +101,7 @@ backend: tokenizer-mode: deepseek_v4 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-high-tpt-megamoe.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-high-tpt-megamoe.yaml index b41d84056b..f2ecf85864 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-high-tpt-megamoe.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-high-tpt-megamoe.yaml @@ -112,7 +112,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-low-latency.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-low-latency.yaml index bd104db522..858bb42c20 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-low-latency.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-low-latency.yaml @@ -109,7 +109,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1x32x64x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-mid-curve-megamoe.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-mid-curve-megamoe.yaml index 807f545b49..60104c28ab 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-mid-curve-megamoe.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-b300-mid-curve-megamoe.yaml @@ -112,7 +112,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "256x1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe-mtp2.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe-mtp2.yaml index 7116999e5b..37fe90fa0a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe-mtp2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe-mtp2.yaml @@ -128,7 +128,7 @@ backend: tokenizer-mode: deepseek_v4 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe.yaml index 2d32da784e..2491911749 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-high-tpt-megamoe.yaml @@ -137,7 +137,7 @@ backend: tokenizer-mode: deepseek_v4 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency-mtp2.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency-mtp2.yaml index 52b64b1675..5ad64359f9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency-mtp2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency-mtp2.yaml @@ -114,7 +114,7 @@ backend: tokenizer-mode: deepseek_v4 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: 16x32x64 diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency.yaml index 423a7ea9b2..0939ef1b4c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-latency.yaml @@ -134,7 +134,7 @@ backend: tokenizer-mode: deepseek_v4 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-middle-curve.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-middle-curve.yaml index c89648d50d..b9644cc1e4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-middle-curve.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-low-middle-curve.yaml @@ -136,7 +136,7 @@ backend: tokenizer-mode: deepseek_v4 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "256x512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-max-tpt-megamoe.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-max-tpt-megamoe.yaml index 92141b5ee1..9abb2dee31 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-max-tpt-megamoe.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-max-tpt-megamoe.yaml @@ -137,7 +137,7 @@ backend: tokenizer-mode: deepseek_v4 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe-mtp2.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe-mtp2.yaml index 521c0db9ff..98fc405785 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe-mtp2.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe-mtp2.yaml @@ -128,7 +128,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "128x256x512x1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe.yaml index 94ec4c4a5e..bf571bdce7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb200-mid-curve-megamoe.yaml @@ -137,7 +137,7 @@ backend: tokenizer-mode: deepseek_v4 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "256x512x1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p6d-dep4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p6d-dep4-tp4.yaml index 577578c340..06a1dd130a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p6d-dep4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p6d-dep4-tp4.yaml @@ -108,7 +108,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p9d-tep4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p9d-tep4-tp4.yaml index c04b3cf078..f0bc4bf734 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p9d-tep4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-1p9d-tep4-tp4.yaml @@ -99,7 +99,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "18" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml index 0ee9a7543b..0b3ac88329 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-4p1d-dep4-dep8-24-c4096.yaml @@ -117,7 +117,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml index ce74045191..cd22158c45 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-5p1d-dep4-dep8-28-c4096.yaml @@ -117,7 +117,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml index 8b98e13bee..9ff033c6ec 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-6p1d-dep4-dep8-32-c4096.yaml @@ -117,7 +117,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-7p2d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-7p2d-dep4-dep16.yaml index 2e118c1c94..c70b580494 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-7p2d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/8k1k/disagg-gb300-7p2d-dep4-dep16.yaml @@ -114,7 +114,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "3072" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep16.yaml index bde0e1a259..2249645140 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep16.yaml @@ -94,7 +94,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "4096x6144" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep8.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep8.yaml index c26c58be30..d627c790bb 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep8.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p1d-dep4-dep8.yaml @@ -94,7 +94,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "4096x12288" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p4d-dep4-tp8.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p4d-dep4-tp8.yaml index 2b813122c9..67bed3b38c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p4d-dep4-tp8.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb200-1p4d-dep4-tp8.yaml @@ -90,7 +90,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "4x8x32x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep16.yaml index 5f5a128572..72eff7bf9f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep16.yaml @@ -93,7 +93,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "2048x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep24.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep24.yaml index 0ccd1766cc..d020dd9b01 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep24.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p1d-dep4-dep24.yaml @@ -93,7 +93,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p2d-dep4-dep4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p2d-dep4-dep4.yaml index 1d1db2fc03..830c70a06c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p2d-dep4-dep4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p2d-dep4-dep4.yaml @@ -93,7 +93,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "6144" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p7d-tep4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p7d-tep4-tp4.yaml index c421bb9ea7..20f2060b14 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p7d-tep4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-1p7d-tep4-tp4.yaml @@ -88,7 +88,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "8x16x32x64x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-2p3d-dep4-dep8.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-2p3d-dep4-dep8.yaml index f157f8700b..8b415ceb2c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-2p3d-dep4-dep8.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/1k1k/disagg-gb300-2p3d-dep4-dep8.yaml @@ -93,7 +93,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tep4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tep4.yaml index 3cf09eb3a8..5e34443e79 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tep4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tep4.yaml @@ -90,7 +90,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tp8.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tp8.yaml index e63fc93db6..c6da14d0c2 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tp8.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-1p4d-dep4-tp8.yaml @@ -90,7 +90,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4x8x16x32x256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-3p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-3p1d-dep4-dep16.yaml index 7c8d113ee5..f4625a2e93 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-3p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-3p1d-dep4-dep16.yaml @@ -93,7 +93,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-6p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-6p1d-dep4-dep16.yaml index 4d6dc7f9c5..022a7f3635 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-6p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-6p1d-dep4-dep16.yaml @@ -93,7 +93,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "3072" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-8p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-8p1d-dep4-dep16.yaml index 634b244471..5e6427b6f9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-8p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb200-8p1d-dep4-dep16.yaml @@ -94,7 +94,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "6144" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p4d-dep4-tp8.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p4d-dep4-tp8.yaml index 6f03f2e5c8..665e0a66ad 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p4d-dep4-tp8.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p4d-dep4-tp8.yaml @@ -90,7 +90,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p8d-dep4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p8d-dep4-tp4.yaml index fef3335660..554b351068 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p8d-dep4-tp4.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-1p8d-dep4-tp4.yaml @@ -90,7 +90,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "32x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-2p1d-dep4-dep24.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-2p1d-dep4-dep24.yaml index 04f7ee4fc3..dbd894d725 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-2p1d-dep4-dep24.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-2p1d-dep4-dep24.yaml @@ -93,7 +93,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-3p1d-dep4-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-3p1d-dep4-dep16.yaml index 37ce4930d0..e399ca0fb5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-3p1d-dep4-dep16.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-3p1d-dep4-dep16.yaml @@ -93,7 +93,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-4p1d-dep4-dep8.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-4p1d-dep4-dep8.yaml index c3caeb63a1..275b5c3137 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-4p1d-dep4-dep8.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-4p1d-dep4-dep8.yaml @@ -94,7 +94,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "3072" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-8p1d-dep4-dep24.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-8p1d-dep4-dep24.yaml index ca322f7c3e..6c4a0e3b21 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-8p1d-dep4-dep24.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k2.5-fp4/8k1k/disagg-gb300-8p1d-dep4-dep24.yaml @@ -94,7 +94,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "15360" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-dep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-dep8-1k1k.yaml index 7b9cd40ca4..b859b058ef 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-dep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-dep8-1k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "1024x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-tep8-1k1k.yaml index 35d814894f..963f9ba4d9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p1d-dep4-tep8-1k1k.yaml @@ -99,7 +99,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "128x256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-dep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-dep8-1k1k.yaml index 9f1e40734f..6134dfc637 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-dep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-dep8-1k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "1024x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-tep8-1k1k.yaml index 0deec8c06a..edae8be6ac 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p2d-dep4-tep8-1k1k.yaml @@ -99,7 +99,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "4x16x64" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p4d-dep4-tp4-marlin-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p4d-dep4-tp4-marlin-1k1k.yaml index 59ab4a338f..fc817c5507 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p4d-dep4-tp4-marlin-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/1k1k/1p4d-dep4-tp4-marlin-1k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "32" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-dep8-8k1k.yaml index 3e6a531cd6..10a1c82ce3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-dep8-8k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep4-8k1k.yaml index 8587b7e852..18b63411b6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep4-8k1k.yaml @@ -99,7 +99,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep8-8k1k.yaml index 15e85eb784..4b9ce2e876 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/1p2d-dep4-tep8-8k1k.yaml @@ -99,7 +99,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4x16x32x64x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/2p2d-dep4-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/2p2d-dep4-dep8-8k1k.yaml index 19f61ccef0..38e183c3fe 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/2p2d-dep4-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/2p2d-dep4-dep8-8k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "512x1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/3p2d-dep4-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/3p2d-dep4-dep8-8k1k.yaml index 3fa2f44ed7..c3aed93778 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/3p2d-dep4-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/3p2d-dep4-dep8-8k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/5p2d-dep4-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/5p2d-dep4-dep8-8k1k.yaml index 4d1413ac2b..0ab797f970 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/5p2d-dep4-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb200-fp8/8k1k/5p2d-dep4-dep8-8k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p1d-dep2-dep4-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p1d-dep2-dep4-1k1k.yaml index a5e3886df9..b7ef8410f1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p1d-dep2-dep4-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p1d-dep2-dep4-1k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p2d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p2d-dep2-tep8-1k1k.yaml index 18aba17e7f..76be752718 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p2d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/1p2d-dep2-tep8-1k1k.yaml @@ -99,7 +99,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "4x16x64x128x256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml index 6b8f278c80..cad7e88dac 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml @@ -99,7 +99,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "32" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p3d-dep2-dep4-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p3d-dep2-dep4-1k1k.yaml index 4ca19873cd..e422be56f0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p3d-dep2-dep4-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p3d-dep2-dep4-1k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p4d-dep2-dep4-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p4d-dep2-dep4-1k1k.yaml index 7449e1ea0c..5f12be921a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p4d-dep2-dep4-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/2p4d-dep2-dep4-1k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "8192" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/4p2d-dep2-dep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/4p2d-dep2-dep8-1k1k.yaml index 61d575b44e..4a00a96e38 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/4p2d-dep2-dep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/1k1k/4p2d-dep2-dep8-1k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "1024x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-dep8-8k1k.yaml index 7ab2a94346..3158ca470c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-dep8-8k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-tep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-tep8-8k1k.yaml index b9da40127a..130a32897f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-tep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p1d-dep2-tep8-8k1k.yaml @@ -99,7 +99,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p2d-dep2-tep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p2d-dep2-tep8-8k1k.yaml index 6f0c19cdca..c7a51865cf 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p2d-dep2-tep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/1p2d-dep2-tep8-8k1k.yaml @@ -99,7 +99,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "32x64x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p1d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p1d-dep2-dep8-8k1k.yaml index f09539dcc4..bd64ee4d32 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p1d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p1d-dep2-dep8-8k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml index b5b40bbdca..daaec01a6e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml @@ -99,7 +99,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p4d-dep2-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p4d-dep2-tep4-8k1k.yaml index 31b5cd3abc..3765bec505 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p4d-dep2-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/2p4d-dep2-tep4-8k1k.yaml @@ -99,7 +99,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p1d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p1d-dep2-dep8-8k1k.yaml index 0dbe0bf5d1..5b31a0f2a6 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p1d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p1d-dep2-dep8-8k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p2d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p2d-dep2-dep8-8k1k.yaml index 3b49f4b1d5..222b5547e5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p2d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/3p2d-dep2-dep8-8k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/6p1d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/6p1d-dep2-dep8-8k1k.yaml index 1abcba608c..61fd8a6f6c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/6p1d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3-gb300-fp8/8k1k/6p1d-dep2-dep8-8k1k.yaml @@ -101,7 +101,7 @@ backend: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tep8-1k1k.yaml index 791fba07d8..2db08084e4 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tep8-1k1k.yaml @@ -75,7 +75,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "4x16x64x128x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tp4-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tp4-1k1k.yaml index 9b568904d3..c840f29f9e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tp4-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p1d-dep2-tp4-1k1k.yaml @@ -76,7 +76,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "1x4x8x16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p2d-dep2-dep4-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p2d-dep2-dep4-1k1k.yaml index 65150ea97f..e3bdf4788a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p2d-dep2-dep4-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/1p2d-dep2-dep4-1k1k.yaml @@ -77,7 +77,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-dep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-dep8-1k1k.yaml index 218da8486b..45bc95a331 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-dep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-dep8-1k1k.yaml @@ -77,7 +77,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "512x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-tep8-1k1k.yaml index e1f1dc3a08..85d6c26e21 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p1d-dep2-tep8-1k1k.yaml @@ -75,7 +75,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "32" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p2d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p2d-dep2-tep8-1k1k.yaml index 12638b008e..6a2d2601eb 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p2d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/2p2d-dep2-tep8-1k1k.yaml @@ -75,7 +75,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/3p2d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/3p2d-dep2-tep8-1k1k.yaml index 0e298026cf..3b531b5d1f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/3p2d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/1k1k/3p2d-dep2-tep8-1k1k.yaml @@ -75,7 +75,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp4-8k1k.yaml index 93eb667668..e120bd1106 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp4-8k1k.yaml @@ -67,7 +67,7 @@ health_check: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "8x16x24x32x48x64" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp8-8k1k.yaml index 83117adcbf..9339941265 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/1p1d-tp1-tp8-8k1k.yaml @@ -66,7 +66,7 @@ health_check: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1x2x4x8x16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/2p1d-tp1-tp4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/2p1d-tp1-tp4-8k1k.yaml index 959cfadc84..8649e04a1a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/2p1d-tp1-tp4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/2p1d-tp1-tp4-8k1k.yaml @@ -67,7 +67,7 @@ health_check: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/3p1d-tp1-tp4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/3p1d-tp1-tp4-8k1k.yaml index 4c9a62bbcc..27fb9a0e1b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/3p1d-tp1-tp4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/3p1d-tp1-tp4-8k1k.yaml @@ -67,7 +67,7 @@ health_check: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "256" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p1d-tp1-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p1d-tp1-tep4-8k1k.yaml index e7d5c27a25..32a04855f7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p1d-tp1-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p1d-tp1-tep4-8k1k.yaml @@ -67,7 +67,7 @@ health_check: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p2d-dep2-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p2d-dep2-tep4-8k1k.yaml index 7c9cec6fa1..95cf8af823 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p2d-dep2-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/4p2d-dep2-tep4-8k1k.yaml @@ -79,7 +79,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/5p2d-tp1-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/5p2d-tp1-dep8-8k1k.yaml index f7f5742a8c..a8090be5d5 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/5p2d-tp1-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/5p2d-tp1-dep8-8k1k.yaml @@ -68,7 +68,7 @@ health_check: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/8p2d-tp1-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/8p2d-tp1-dep8-8k1k.yaml index a5d8a60755..e089eec422 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/8p2d-tp1-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/8p2d-tp1-dep8-8k1k.yaml @@ -68,7 +68,7 @@ health_check: interval_seconds: 10 benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "768x1024" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p1d-dep2-tp4-eagle3-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p1d-dep2-tp4-eagle3-8k1k.yaml index b6a84ee773..2fdb0701d3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p1d-dep2-tp4-eagle3-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p1d-dep2-tp4-eagle3-8k1k.yaml @@ -83,7 +83,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p2d-dep2-tp4-eagle3-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p2d-dep2-tp4-eagle3-8k1k.yaml index 20636eaa93..564b8a214a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p2d-dep2-tp4-eagle3-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p2d-dep2-tp4-eagle3-8k1k.yaml @@ -83,7 +83,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p4d-dep2-tp4-eagle3-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p4d-dep2-tp4-eagle3-8k1k.yaml index 790694a545..1877bb1569 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p4d-dep2-tp4-eagle3-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p4d-dep2-tp4-eagle3-8k1k.yaml @@ -83,7 +83,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p6d-dep2-tp4-eagle3-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p6d-dep2-tp4-eagle3-8k1k.yaml index 7973dd2bdf..333802b334 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p6d-dep2-tp4-eagle3-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/1p6d-dep2-tp4-eagle3-8k1k.yaml @@ -83,7 +83,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p1d-dep2-dep4-eagle3-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p1d-dep2-dep4-eagle3-8k1k.yaml index da3ad68609..3a6635f992 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p1d-dep2-dep4-eagle3-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p1d-dep2-dep4-eagle3-8k1k.yaml @@ -87,7 +87,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p3d-dep2-tp4-eagle3-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p3d-dep2-tp4-eagle3-8k1k.yaml index 96ecdbefc2..477b305010 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p3d-dep2-tp4-eagle3-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp4/8k1k/mtp/2p3d-dep2-tp4-eagle3-8k1k.yaml @@ -83,7 +83,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 req_rate: inf diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tep8-1k1k.yaml index c9c2cf06b2..e61b17525b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tep8-1k1k.yaml @@ -73,7 +73,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "4x16x64x128x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tp4-marlin-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tp4-marlin-1k1k.yaml index d635811972..b012823f69 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tp4-marlin-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p1d-dep2-tp4-marlin-1k1k.yaml @@ -75,7 +75,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "1x4x8x16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p2d-dep2-dep4-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p2d-dep2-dep4-1k1k.yaml index c8dfbecbee..1228e5863f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p2d-dep2-dep4-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/1p2d-dep2-dep4-1k1k.yaml @@ -75,7 +75,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "2048" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-dep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-dep8-1k1k.yaml index 5596247b22..d3c1e0a33e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-dep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-dep8-1k1k.yaml @@ -75,7 +75,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "512x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-tep8-1k1k.yaml index 8882b0fc92..72893e968f 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p1d-dep2-tep8-1k1k.yaml @@ -73,7 +73,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "32" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml index 9b646e1d3a..668eb5289a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/2p2d-dep2-tep8-1k1k.yaml @@ -73,7 +73,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/3p2d-dep2-tep8-1k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/3p2d-dep2-tep8-1k1k.yaml index c6bfd28bc6..23bf5abf4e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/3p2d-dep2-tep8-1k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/1k1k/3p2d-dep2-tep8-1k1k.yaml @@ -73,7 +73,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 1024 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p1d-dep2-tp4-marlin-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p1d-dep2-tp4-marlin-8k1k.yaml index b417471c89..1b1959900d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p1d-dep2-tp4-marlin-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p1d-dep2-tp4-marlin-8k1k.yaml @@ -79,7 +79,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "1x4x8x16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p2d-dep2-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p2d-dep2-tep4-8k1k.yaml index 849c81fb85..7004f72ff8 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p2d-dep2-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p2d-dep2-tep4-8k1k.yaml @@ -77,7 +77,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep4-8k1k.yaml index 736d23398f..fdfb00c4e3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep4-8k1k.yaml @@ -77,7 +77,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "16x32x64x128" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep8-8k1k.yaml index 01c414f4fc..56e4f0b4ad 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/1p4d-dep2-tep8-8k1k.yaml @@ -77,7 +77,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-dep8-8k1k.yaml index 9326a9301f..6bd7ffbfa0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-dep8-8k1k.yaml @@ -79,7 +79,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "256x512" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml index e289ce7949..8e2e70baa3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/2p2d-dep2-tep8-8k1k.yaml @@ -77,7 +77,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-dep8-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-dep8-8k1k.yaml index 49c57c199f..1f9b3c759b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-dep8-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-dep8-8k1k.yaml @@ -79,7 +79,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-tep4-8k1k.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-tep4-8k1k.yaml index 2a7faba26d..69710288e3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-tep4-8k1k.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/minimax-m3/b300-fp8/8k1k/4p2d-dep2-tep4-8k1k.yaml @@ -77,7 +77,7 @@ health_check: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh + command: bash /infmax-workspace/benchmarks/multi_node/srt_bench_serving_entrypoint.sh isl: 8192 osl: 1024 concurrencies: "4096" diff --git a/benchmarks/multi_node/srt_bench.sh b/benchmarks/multi_node/srt_bench.sh deleted file mode 100755 index 7d7466f0cc..0000000000 --- a/benchmarks/multi_node/srt_bench.sh +++ /dev/null @@ -1,80 +0,0 @@ -#!/usr/bin/env bash - -# InferenceX benchmark_serving adapter for srt-slurm's custom benchmark runner. -# The recipe hydrator in runners/srt_slurm_benchmark_config.py supplies the -# environment below from the selected recipe's throughput fields. - -set -euo pipefail - -INFMAX_WS="${INFMAX_CONTAINER_WORKSPACE:-/infmax-workspace}" -# shellcheck disable=SC1091 -source "$INFMAX_WS/benchmarks/benchmark_lib.sh" - -check_env_vars MODEL TOKENIZER ISL OSL CONC_LIST REQ_RATE DISAGG \ - PREFILL_GPUS DECODE_GPUS TOTAL_GPUS - -BENCH_HOST="${SRT_FRONTEND_HOST:-127.0.0.1}" -BENCH_PORT="${SRT_FRONTEND_PORT:-8000}" -RANDOM_RANGE_RATIO="${RANDOM_RANGE_RATIO:-0.8}" -NUM_PROMPTS_MULT="${NUM_PROMPTS_MULT:-10}" -NUM_WARMUP_MULT="${NUM_WARMUP_MULT:-2}" -USE_CHAT_TEMPLATE="${USE_CHAT_TEMPLATE:-true}" -DSV4="${DSV4:-false}" -TRUST_REMOTE_CODE="${TRUST_REMOTE_CODE:-true}" -RESULT_DIR="/logs/inferencex-bench_isl_${ISL}_osl_${OSL}" - -ensure_bench_serving_deps() { - local deps=(aiohttp numpy tqdm transformers huggingface_hub) - if python3 -c \ - "import aiohttp, numpy, tqdm, transformers, huggingface_hub" \ - 2>/dev/null; then - return - fi - - local venv="/tmp/inferencex-srt-bench-venv" - [[ -d "$venv" ]] || python3 -m venv --system-site-packages "$venv" - # shellcheck disable=SC1091 - source "$venv/bin/activate" - pip install --quiet "${deps[@]}" -} - -ensure_bench_serving_deps -mkdir -p "$RESULT_DIR" - -curl -fsS "http://${BENCH_HOST}:${BENCH_PORT}/v1/models" >/dev/null || { - echo "InferenceX benchmark could not reach the srt-slurm frontend" >&2 - exit 66 -} -ulimit -n 65536 2>/dev/null || true - -read -r -a concurrencies <<< "${CONC_LIST//x/ }" -for concurrency in "${concurrencies[@]}"; do - if [[ "$DISAGG" == "true" ]]; then - result_filename="results_concurrency_${concurrency}_gpus_${TOTAL_GPUS}_ctx_${PREFILL_GPUS}_gen_${DECODE_GPUS}" - else - result_filename="results_concurrency_${concurrency}_gpus_${TOTAL_GPUS}" - fi - - args=( - --model "$MODEL" - --tokenizer "$TOKENIZER" - --host "$BENCH_HOST" - --port "$BENCH_PORT" - --backend openai - --input-len "$ISL" - --output-len "$OSL" - --random-range-ratio "$RANDOM_RANGE_RATIO" - --num-prompts "$((concurrency * NUM_PROMPTS_MULT))" - --num-warmups "$((concurrency * NUM_WARMUP_MULT))" - --request-rate "$REQ_RATE" - --max-concurrency "$concurrency" - --result-filename "$result_filename" - --result-dir "$RESULT_DIR" - --bench-serving-dir "$INFMAX_WS" - ) - [[ "$USE_CHAT_TEMPLATE" == "true" ]] && args+=(--use-chat-template) - [[ "$DSV4" == "true" ]] && args+=(--dsv4) - [[ "$TRUST_REMOTE_CODE" == "true" ]] && args+=(--trust-remote-code) - - run_benchmark_serving "${args[@]}" -done diff --git a/benchmarks/multi_node/srt_bench_serving_entrypoint.sh b/benchmarks/multi_node/srt_bench_serving_entrypoint.sh new file mode 100755 index 0000000000..89e9004a5e --- /dev/null +++ b/benchmarks/multi_node/srt_bench_serving_entrypoint.sh @@ -0,0 +1,94 @@ +#!/usr/bin/env bash + +# InferenceX benchmark_serving adapter for srt-slurm's custom benchmark runner. + +set -eo pipefail + +if [[ $# -ne 15 ]]; then + echo "Expected 15 benchmark arguments, received $#" >&2 + exit 64 +fi + +readonly MODEL=$1 +readonly TOKENIZER=$2 +readonly ISL=$3 +readonly OSL=$4 +readonly CONC_LIST=$5 +readonly REQ_RATE=$6 +readonly DISAGG=$7 +readonly PREFILL_GPUS=$8 +readonly DECODE_GPUS=$9 +readonly TOTAL_GPUS=${10} +readonly RANDOM_RANGE_RATIO=${11} +readonly NUM_PROMPTS_MULT=${12} +readonly NUM_WARMUP_MULT=${13} +readonly USE_CHAT_TEMPLATE=${14} +readonly DSV4=${15} + +readonly INFERENCEX_WORKSPACE=/infmax-workspace +readonly BENCH_HOST=127.0.0.1 +readonly BENCH_PORT=8000 +readonly BENCH_BACKEND=openai +readonly TRUST_REMOTE_CODE=true +readonly RESULT_DIR="/logs/inferencex-bench_isl_${ISL}_osl_${OSL}" +export EVAL_ONLY=false +export PROFILE=0 + +# shellcheck disable=SC1091 +source "$INFERENCEX_WORKSPACE/benchmarks/benchmark_lib.sh" + +ensure_bench_serving_deps() { + local deps=(aiohttp numpy tqdm transformers huggingface_hub) + if python3 -c \ + "import aiohttp, numpy, tqdm, transformers, huggingface_hub" \ + 2>/dev/null; then + return + fi + + local venv="/tmp/inferencex-srt-bench-venv" + [[ -d "$venv" ]] || python3 -m venv --system-site-packages "$venv" + # shellcheck disable=SC1091 + source "$venv/bin/activate" + pip install --quiet "${deps[@]}" +} + +ensure_bench_serving_deps +mkdir -p "$RESULT_DIR" + +curl -fsS "http://${BENCH_HOST}:${BENCH_PORT}/v1/models" >/dev/null || { + echo "InferenceX benchmark could not reach the srt-slurm frontend" >&2 + exit 66 +} +ulimit -n 65536 2>/dev/null || true + +read -r -a concurrencies <<< "${CONC_LIST//x/ }" +for concurrency in "${concurrencies[@]}"; do + if [[ "$DISAGG" == "true" ]]; then + result_filename="results_concurrency_${concurrency}_gpus_${TOTAL_GPUS}_ctx_${PREFILL_GPUS}_gen_${DECODE_GPUS}" + else + result_filename="results_concurrency_${concurrency}_gpus_${TOTAL_GPUS}" + fi + + optional_args=() + [[ "$USE_CHAT_TEMPLATE" == "true" ]] && optional_args+=(--use-chat-template) + [[ "$DSV4" == "true" ]] && optional_args+=(--dsv4) + [[ "$TRUST_REMOTE_CODE" == "true" ]] && optional_args+=(--trust-remote-code) + + run_benchmark_serving \ + --model "$MODEL" \ + --tokenizer "$TOKENIZER" \ + --host "$BENCH_HOST" \ + --port "$BENCH_PORT" \ + --backend "$BENCH_BACKEND" \ + --input-len "$ISL" \ + --output-len "$OSL" \ + --random-range-ratio "$RANDOM_RANGE_RATIO" \ + --num-prompts "$((concurrency * NUM_PROMPTS_MULT))" \ + --num-warmups "$((concurrency * NUM_WARMUP_MULT))" \ + --request-rate "$REQ_RATE" \ + --max-concurrency "$concurrency" \ + --result-filename "$result_filename" \ + --result-dir "$RESULT_DIR" \ + --bench-serving-dir "$INFERENCEX_WORKSPACE" \ + "${optional_args[@]}" +done diff --git a/runners/srt_slurm_benchmark_config.py b/runners/srt_slurm_benchmark_config.py index 202d65ed77..06d0f52856 100755 --- a/runners/srt_slurm_benchmark_config.py +++ b/runners/srt_slurm_benchmark_config.py @@ -5,6 +5,7 @@ import argparse import hashlib +import shlex from pathlib import Path from typing import Any @@ -18,9 +19,11 @@ from srtctl.core.schema import SrtConfig -CUSTOM_BENCHMARK_COMMAND = ( - "bash /infmax-workspace/benchmarks/multi_node/srt_bench.sh" +CUSTOM_BENCHMARK_ENTRYPOINT = ( + "/infmax-workspace/benchmarks/multi_node/" + "srt_bench_serving_entrypoint.sh" ) +CUSTOM_BENCHMARK_COMMAND = f"bash {CUSTOM_BENCHMARK_ENTRYPOINT}" def split_config_spec(config_spec: str) -> tuple[Path, str | None]: @@ -88,7 +91,7 @@ def load_selected_recipes( def stringify_concurrencies(concurrencies: list[int] | str | None) -> str: - """Return the whitespace-separated format consumed by srt_bench.sh.""" + """Return the entrypoint's whitespace-separated concurrency format.""" if isinstance(concurrencies, list): return " ".join(str(value) for value in concurrencies) if concurrencies is None: @@ -96,8 +99,8 @@ def stringify_concurrencies(concurrencies: list[int] | str | None) -> str: return str(concurrencies).replace("x", " ") -def custom_benchmark_environment(config: SrtConfig) -> dict[str, str]: - """Translate recipe settings to the custom wrapper contract.""" +def custom_benchmark_arguments(config: SrtConfig) -> list[str]: + """Translate recipe settings to entrypoint arguments.""" benchmark = config.benchmark resources = config.resources @@ -143,30 +146,29 @@ def custom_benchmark_environment(config: SrtConfig) -> dict[str, str]: custom_tokenizer = benchmark.custom_tokenizer or "" is_deepseek_v4 = "deepseek_v4" in custom_tokenizer.lower() - return { - "MODEL": config.served_model_name, - "TOKENIZER": tokenizer, - "ISL": str(benchmark.isl), - "OSL": str(benchmark.osl), - "CONC_LIST": concurrencies, - "REQ_RATE": str(benchmark.req_rate or "inf"), - "DISAGG": str(resources.is_disaggregated).lower(), - "PREFILL_GPUS": str(prefill_gpus), - "DECODE_GPUS": str(decode_gpus), - "TOTAL_GPUS": str(total_gpus), - "RANDOM_RANGE_RATIO": str(benchmark.random_range_ratio or 0.8), - "NUM_PROMPTS_MULT": str(benchmark.num_prompts_mult or 10), - "NUM_WARMUP_MULT": str( + return [ + config.served_model_name, + tokenizer, + str(benchmark.isl), + str(benchmark.osl), + concurrencies, + str(benchmark.req_rate or "inf"), + str(resources.is_disaggregated).lower(), + str(prefill_gpus), + str(decode_gpus), + str(total_gpus), + str(benchmark.random_range_ratio or 0.8), + str(benchmark.num_prompts_mult or 10), + str( benchmark.num_warmup_mult if benchmark.num_warmup_mult is not None else 2 ), - "USE_CHAT_TEMPLATE": str(benchmark.use_chat_template).lower(), - "DSV4": str( + str(benchmark.use_chat_template).lower(), + str( is_deepseek_v4 and benchmark.use_chat_template ).lower(), - "TRUST_REMOTE_CODE": "true", - } + ] def hydrate_recipe(recipe: dict[str, Any]) -> bool: @@ -191,9 +193,14 @@ def hydrate_recipe(recipe: dict[str, Any]) -> bool: if not isinstance(config, SrtConfig): raise TypeError("srt-slurm did not return a typed SrtConfig") - environment = dict(benchmark.get("env") or {}) - environment.update(custom_benchmark_environment(config)) - benchmark["env"] = environment + benchmark["command"] = shlex.join( + [ + "bash", + CUSTOM_BENCHMARK_ENTRYPOINT, + *custom_benchmark_arguments(config), + ] + ) + benchmark.pop("env", None) return True