diff --git a/benchmarks/multi_node/srt-slurm-recipe-identities.yaml b/benchmarks/multi_node/srt-slurm-recipe-identities.yaml index 6b4ccaa1c8..39028027f0 100644 --- a/benchmarks/multi_node/srt-slurm-recipe-identities.yaml +++ b/benchmarks/multi_node/srt-slurm-recipe-identities.yaml @@ -1,3 +1,365 @@ # CONFIG_FILE selectors that replaced flat recipes, mapped to the replaced # CONFIG_FILE. Matrix fingerprints and curve identity use the replaced path. -{} +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp8_dp8_d_tp8_ep8_dp8_b256_c160x288_mtp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p1d-p-tp8-dp8-d-tp8-ep8-dp8-b256-c160x288-mtp.yaml +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp8_dp8_d_tp8_ep8_dp8_b256_c160x288_stp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p1d-p-tp8-dp8-d-tp8-ep8-dp8-b256-c160x288-stp.yaml +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p2d_p_tp8_dp8_d_tp8_ep8_dp8_b128_c288_mtp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p2d-p-tp8-dp8-d-tp8-ep8-dp8-b128-c288-mtp.yaml +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p2d_p_tp8_dp8_d_tp8_ep8_dp8_b128_c288_stp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p2d-p-tp8-dp8-d-tp8-ep8-dp8-b128-c288-stp.yaml +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p3d_p_tp8_dp8_d_tp8_b32_c128_mtp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p3d-p-tp8-dp8-d-tp8-b32-c128-mtp.yaml +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p3d_p_tp8_dp8_d_tp8_b32_c128_stp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p3d-p-tp8-dp8-d-tp8-b32-c128-stp.yaml +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p4d_p_tp8_dp8_d_tp8_b32_c128_mtp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p4d-p-tp8-dp8-d-tp8-b32-c128-mtp.yaml +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p4d_p_tp8_dp8_d_tp8_b32_c128_stp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p4d-p-tp8-dp8-d-tp8-b32-c128-stp.yaml +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p6d_p_tp8_dp8_d_tp8_b22_c8x16x32x64x128_mtp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p6d-p-tp8-dp8-d-tp8-b22-c8x16x32x64x128-mtp.yaml +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p6d_p_tp8_dp8_d_tp8_b22_c8x16x32x64x128_stp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p6d-p-tp8-dp8-d-tp8-b22-c8x16x32x64x128-stp.yaml +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_2p1d_p_tp8_dp8_d_tp8_ep8_dp8_b512_c512_mtp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-2p1d-p-tp8-dp8-d-tp8-ep8-dp8-b512-c512-mtp.yaml +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_2p1d_p_tp8_dp8_d_tp8_ep8_dp8_b512_c512_stp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-2p1d-p-tp8-dp8-d-tp8-ep8-dp8-b512-c512-stp.yaml +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_3p1d_p_tp8_dp8_d_tp8_ep8_dp8_b1024_c1024_mtp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-3p1d-p-tp8-dp8-d-tp8-ep8-dp8-b1024-c1024-mtp.yaml +recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_3p1d_p_tp8_dp8_d_tp8_ep8_dp8_b1024_c1024_stp: recipes/dsr1/sglang/b200-fp8/8k1k/disagg-3p1d-p-tp8-dp8-d-tp8-ep8-dp8-b1024-c1024-stp.yaml +recipes/dsr1/sglang/gb200-fp4/8k1k/variants.yaml:override_disagg_10p1d_p_tp4_d_tp32_ep32_dp32_b16384_c2048_stp: recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-10p1d-p-tp4-d-tp32-ep32-dp32-b16384-c2048-stp.yaml +recipes/dsr1/sglang/gb200-fp4/8k1k/variants.yaml:override_disagg_1p4d_p_tp4_d_tp4_c4x8_stp: recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-1p4d-p-tp4-d-tp4-c4x8-stp.yaml +recipes/dsr1/sglang/gb200-fp4/8k1k/variants.yaml:override_disagg_6p1d_p_tp4_d_tp48_ep48_dp48_b16384_c512x2048x4096_stp: recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-6p1d-p-tp4-d-tp48-ep48-dp48-b16384-c512x2048x4096-stp.yaml +? recipes/dsr1/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_5p1d_p_tp8_ep8_dp8_d_tp32_ep32_dp32_b8192_c512x1024x2048x6144_stp +: recipes/dsr1/sglang/gb200-fp8/8k1k/disagg-5p1d-p-tp8-ep8-dp8-d-tp32-ep32-dp32-b8192-c512x1024x2048x6144-stp.yaml +? recipes/dsr1/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_6p1d_p_tp8_ep8_dp8_d_tp24_ep24_dp24_b8192_c2048x4096x6144_stp +: recipes/dsr1/sglang/gb200-fp8/8k1k/disagg-6p1d-p-tp8-ep8-dp8-d-tp24-ep24-dp24-b8192-c2048x4096x6144-stp.yaml +recipes/dsr1/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_10p1d_p_tp4_d_tp32_ep32_dp32_b16384_c2048_stp: recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-10p1d-p-tp4-d-tp32-ep32-dp32-b16384-c2048-stp.yaml +recipes/dsr1/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_p_tp4_d_tp4_c4x8x32x64_stp: recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-1p4d-p-tp4-d-tp4-c4x8x32x64-stp.yaml +recipes/dsr1/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_6p1d_p_tp4_d_tp48_ep48_dp48_b16384_c512x2048x4096_stp: recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-6p1d-p-tp4-d-tp48-ep48-dp48-b16384-c512x2048x4096-stp.yaml +recipes/dsr1/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_b128_c4x8_stp: recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c4x8-stp.yaml +? recipes/dsr1/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_5p1d_p_tp8_ep8_dp8_d_tp32_ep32_dp32_b45000_c128x256x512x1024_stp +: recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-5p1d-p-tp8-ep8-dp8-d-tp32-ep32-dp32-b45000-c128x256x512x1024-stp.yaml +recipes/dsr1/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_6p1d_p_tp8_ep8_dp8_d_tp24_ep24_dp24_b45000_c2048x4096_stp: recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp8-ep8-dp8-d-tp24-ep24-dp24-b45000-c2048x4096-stp.yaml +recipes/dsr1/sglang/h100-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp16_d_tp16_b128_c1x2x4x8x16x32x64x128_mtp: recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-b128-c1x2x4x8x16x32x64x128-mtp.yaml +recipes/dsr1/sglang/h100-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp16_d_tp16_b128_c1x2x4x8x16x32x64x128_stp: recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-b128-c1x2x4x8x16x32x64x128-stp.yaml +recipes/dsr1/sglang/h100-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp16_d_tp16_ep16_dp16_b64_c1x2x4x8x16x32x64_mtp: recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-ep16-dp16-b64-c1x2x4x8x16x32x64-mtp.yaml +recipes/dsr1/sglang/h100-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp16_d_tp16_ep16_dp16_b64_c1x2x4x8x16x32x64_stp: recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-ep16-dp16-b64-c1x2x4x8x16x32x64-stp.yaml +recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs128_1p1d_dep_mtp: recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs128-1p1d-dep-mtp.yaml +recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs128_1p1d_dep_stp: recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs128-1p1d-dep-stp.yaml +recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs16_1p3d_mtp: recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs16-1p3d-mtp.yaml +recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs16_1p3d_stp: recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs16-1p3d-stp.yaml +recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs4_1p7d_mtp: recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs4-1p7d-mtp.yaml +recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs4_1p7d_stp: recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs4-1p7d-stp.yaml +recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs64_2p3d_mtp: recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs64-2p3d-mtp.yaml +recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs64_2p3d_stp: recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs64-2p3d-stp.yaml +recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs8_1p6d_mtp: recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs8-1p6d-mtp.yaml +recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs8_1p6d_stp: recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs8-1p6d-stp.yaml +recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_1p1d_dep8_b8_eplb0_mtp3: recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p1d-dep8-b8-eplb0-mtp3.yaml +recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_1p3d_tep8_b16_eplb0_mtp3: recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp3.yaml +recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_1p5d_tep8_b16_eplb0_mtp0: recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b16-eplb0-mtp0.yaml +recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_1p5d_tep8_b1_eplb0_mtp0: recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b1-eplb0-mtp0.yaml +recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_1p5d_tep8_b1_eplb0_mtp3: recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b1-eplb0-mtp3.yaml +recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_1p5d_tep8_b8_eplb0_mtp3: recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b8-eplb0-mtp3.yaml +recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_2p5d_tep8_b64_eplb0_mtp0: recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-2p5d-tep8-b64-eplb0-mtp0.yaml +recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_3p1d_dep8_b64_eplb0_mtp3: recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp3.yaml +recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_4p1d_dep8_b192_eplb0_mtp0: recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-4p1d-dep8-b192-eplb0-mtp0.yaml +recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_4p3d_dep8_b32_eplb0_mtp0: recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-4p3d-dep8-b32-eplb0-mtp0.yaml +recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_5p1d_dep8_b192_eplb0_mtp1: recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-5p1d-dep8-b192-eplb0-mtp1.yaml +recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_5p2d_dep8_b32_eplb0_mtp3: recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-5p2d-dep8-b32-eplb0-mtp3.yaml +recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_7p2d_dep8_b128_eplb0_mtp0: recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-7p2d-dep8-b128-eplb0-mtp0.yaml +recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_10p1d_dep8_b256_eplb0_mtp1: recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp1.yaml +recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p3d_tep4_b32_eplb0_mtp0: recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep4-b32-eplb0-mtp0.yaml +recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p3d_tep8_b16_eplb0_mtp0: recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0.yaml +recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p3d_tep8_b1_eplb0_mtp0: recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep8-b1-eplb0-mtp0.yaml +recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep4_b2_eplb0_mtp0: recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep4-b2-eplb0-mtp0.yaml +recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep4_b8_eplb0_mtp3: recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep4-b8-eplb0-mtp3.yaml +recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp3: recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3.yaml +recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b4_eplb0_mtp3: recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3.yaml +recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_3p1d_dep8_b16_eplb0_mtp3: recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-3p1d-dep8-b16-eplb0-mtp3.yaml +recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_5p2d_dep8_b32_eplb0_mtp0: recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-5p2d-dep8-b32-eplb0-mtp0.yaml +recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_6p1d_dep8_b128_eplb0_mtp0: recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-6p1d-dep8-b128-eplb0-mtp0.yaml +recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_8p1d_dep8_b256_eplb0_mtp0: recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-8p1d-dep8-b256-eplb0-mtp0.yaml +recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_9p1d_dep8_b128_eplb0_mtp1: recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-9p1d-dep8-b128-eplb0-mtp1.yaml +recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_1p1d_dp8_b8_eplb0_mtp3_c72: recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p1d-dp8-b8-eplb0-mtp3-c72.yaml +recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_1p2d_tp8_b16_eplb0_mtp3_c40: recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p2d-tp8-b16-eplb0-mtp3-c40.yaml +recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tp8_b16_eplb0_mtp0_c64: recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b16-eplb0-mtp0-c64.yaml +recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tp8_b1_eplb0_mtp3_c8: recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b1-eplb0-mtp3-c8.yaml +recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tp8_b4_eplb0_mtp3_c20: recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b4-eplb0-mtp3-c20.yaml +recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_1p8d_tp8_b1_eplb0_mtp0_c16: recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p8d-tp8-b1-eplb0-mtp0-c16.yaml +recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_2p1d_dp8_b16_eplb0_mtp3_c144: recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-2p1d-dp8-b16-eplb0-mtp3-c144.yaml +recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_2p1d_dp8_b32_eplb0_mtp0_c256: recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-2p1d-dp8-b32-eplb0-mtp0-c256.yaml +recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_3p1d_dp8_b64_eplb0_mtp0_c512: recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-3p1d-dp8-b64-eplb0-mtp0-c512.yaml +recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_3p5d_tp8_b64_eplb0_mtp0_c256: recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-3p5d-tp8-b64-eplb0-mtp0-c256.yaml +recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_4p1d_dp8_b64_eplb0_mtp2_c512: recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-4p1d-dp8-b64-eplb0-mtp2-c512.yaml +recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_5p1d_dp8_b128_eplb0_mtp0_c1075: recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-5p1d-dp8-b128-eplb0-mtp0-c1075.yaml +recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_7p1d_dep8_b384_eplb0_mtp0_c3072: recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-7p1d-dep8-b384-eplb0-mtp0-c3072.yaml +recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_10p1d_dep16_b256_eplb256_mtp0: recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-10p1d-dep16-b256-eplb256-mtp0.yaml +recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_11p1d_dep16_b256_eplb256_mtp1: recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-11p1d-dep16-b256-eplb256-mtp1.yaml +recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b16_eplb0_mtp0: recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b16-eplb0-mtp0.yaml +recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp0: recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0.yaml +recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b8_eplb0_mtp3: recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b8-eplb0-mtp3.yaml +recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_2p1d_dep32_b8_eplb0_mtp0: recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-2p1d-dep32-b8-eplb0-mtp0.yaml +recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_3p1d_dep32_b4_eplb0_mtp3: recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-3p1d-dep32-b4-eplb0-mtp3.yaml +recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_7p1d_dep16_b64_eplb256_mtp1: recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-7p1d-dep16-b64-eplb256-mtp1.yaml +recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_7p1d_dep32_b32_eplb0_mtp0: recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-7p1d-dep32-b32-eplb0-mtp0.yaml +recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_8p1d_dep16_b128_eplb0_mtp0: recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-8p1d-dep16-b128-eplb0-mtp0.yaml +recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_8p1d_dep32_b16_eplb0_mtp3: recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-8p1d-dep32-b16-eplb0-mtp3.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b16_eplb0_mtp0_c63: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0-c63.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b1_eplb0_mtp0_c6: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b1-eplb0-mtp0-c6.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b2_eplb0_mtp3_c6: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b2-eplb0-mtp3-c6.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b4_eplb0_mtp0_c18: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b4-eplb0-mtp0-c18.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b4_eplb0_mtp3_c15: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b4-eplb0-mtp3-c15.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_2p1d_dep32_b2_eplb0_mtp3_c90: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-2p1d-dep32-b2-eplb0-mtp3-c90.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_2p1d_dep32_b8_eplb0_mtp0_c333: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-2p1d-dep32-b8-eplb0-mtp0-c333.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_3p1d_dep16_b16_eplb0_mtp3_c333: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep16-b16-eplb0-mtp3-c333.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_3p1d_dep16_b32_eplb0_mtp0_c615: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep16-b32-eplb0-mtp0-c615.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_3p1d_dep8_b64_eplb0_mtp3_c666: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp3-c666.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_4p1d_dep32_b16_eplb0_mtp0_c666: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-4p1d-dep32-b16-eplb0-mtp0-c666.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_4p1d_dep32_b8_eplb0_mtp3_c333: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-4p1d-dep32-b8-eplb0-mtp3-c333.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_5p1d_dep16_b32_eplb0_mtp3_c666: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-5p1d-dep16-b32-eplb0-mtp3-c666.yaml +recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_5p1d_dep16_b64_eplb0_mtp0_c1229: recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-5p1d-dep16-b64-eplb0-mtp0-c1229.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_10p1d_dep16_b32_eplb0_mtp3: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep16-b32-eplb0-mtp3.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_10p1d_dep8_b256_eplb0_mtp1: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp1.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_11p3d_dep4_b256_eplb0_mtp0: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-11p3d-dep4-b256-eplb0-mtp0.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_13p1d_dep16_b64_eplb256_mtp3: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-13p1d-dep16-b64-eplb256-mtp3.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_14p1d_dep16_b128_eplb0_mtp0: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-14p1d-dep16-b128-eplb0-mtp0.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p3d_tep8_b16_eplb0_mtp0: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p3d_tep8_b8_eplb0_mtp3: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p3d-tep8-b8-eplb0-mtp3.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp0: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp3: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b2_eplb0_mtp0: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b2-eplb0-mtp0.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b4_eplb0_mtp3: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p5d_tep4_b4_eplb0_mtp0: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p5d-tep4-b4-eplb0-mtp0.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_4p1d_dep32_b4_eplb0_mtp3: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-4p1d-dep32-b4-eplb0-mtp3.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_7p1d_dep32_b16_eplb0_mtp0: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep32-b16-eplb0-mtp0.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_8p1d_dep32_b8_eplb0_mtp3: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-8p1d-dep32-b8-eplb0-mtp3.yaml +recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_9p1d_dep16_b64_eplb0_mtp0: recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-9p1d-dep16-b64-eplb0-mtp0.yaml +recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_10p1d_dep16_b64_eplb0_mtp1_c1229: recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-10p1d-dep16-b64-eplb0-mtp1-c1229.yaml +recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp0_c4: recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0-c4.yaml +recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp3_c8: recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3-c8.yaml +recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tep8_b4_eplb0_mtp0_c24: recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp0-c24.yaml +recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tep8_b4_eplb0_mtp3_c24: recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3-c24.yaml +recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tep8_b8_eplb0_mtp0_c36: recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b8-eplb0-mtp0-c36.yaml +recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_4p1d_dep16_b32_eplb0_mtp0_c666: recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-4p1d-dep16-b32-eplb0-mtp0-c666.yaml +recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_6p1d_dep32_b16_eplb0_mtp0_c512: recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-6p1d-dep32-b16-eplb0-mtp0-c512.yaml +recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_6p1d_dep32_b8_eplb0_mtp3_c333: recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-6p1d-dep32-b8-eplb0-mtp3-c333.yaml +recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_7p1d_dep16_b64_eplb0_mtp0_c1229: recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep16-b64-eplb0-mtp0-c1229.yaml +recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_7p1d_dep8_b128_eplb0_mtp1_c1229: recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep8-b128-eplb0-mtp1-c1229.yaml +recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_7p1d_dep8_b256_eplb0_mtp0_c2151: recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep8-b256-eplb0-mtp0-c2151.yaml +recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_8p1d_dep16_b32_eplb0_mtp3_c666: recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-8p1d-dep16-b32-eplb0-mtp3-c666.yaml +recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p1d_dep16_b4_eplb0_mtp3: recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p1d-dep16-b4-eplb0-mtp3.yaml +recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p2d_tep16_b32_eplb0_mtp3: recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p2d-tep16-b32-eplb0-mtp3.yaml +recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p2d_tep16_b64_eplb0_mtp0: recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p2d-tep16-b64-eplb0-mtp0.yaml +recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep16_b1_eplb0_mtp0: recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b1-eplb0-mtp0.yaml +recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep16_b1_eplb0_mtp3: recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b1-eplb0-mtp3.yaml +recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep16_b2_eplb0_mtp0: recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b2-eplb0-mtp0.yaml +recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep16_b2_eplb0_mtp3: recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b2-eplb0-mtp3.yaml +recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep16_b8_eplb0_mtp0: recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b8-eplb0-mtp0.yaml +recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep16_b8_eplb0_mtp3: recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b8-eplb0-mtp3.yaml +recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_2p1d_dep16_b16_eplb0_mtp0: recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-2p1d-dep16-b16-eplb0-mtp0.yaml +recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_2p1d_dep16_b8_eplb0_mtp3: recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-2p1d-dep16-b8-eplb0-mtp3.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p1d_dep8_b256_eplb0_mtp0_c128: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p1d-dep8-b256-eplb0-mtp0-c128.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p1d_dep8_b32_eplb0_mtp2_c64: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p1d-dep8-b32-eplb0-mtp2-c64.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b32_eplb0_mtp0_c48: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p3d-tep8-b32-eplb0-mtp0-c48.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b32_eplb0_mtp2_c48: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p3d-tep8-b32-eplb0-mtp2-c48.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p6d_tep8_b16_eplb0_mtp0_c48: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p6d-tep8-b16-eplb0-mtp0-c48.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p6d_tep8_b32_eplb0_mtp3_c48: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p6d-tep8-b32-eplb0-mtp3-c48.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p7d_tep8_b1_eplb0_mtp0_c9: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b1-eplb0-mtp0-c9.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p7d_tep8_b1_eplb0_mtp3_c9: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b1-eplb0-mtp3-c9.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p7d_tep8_b32_eplb0_mtp0_c28: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b32-eplb0-mtp0-c28.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p7d_tep8_b32_eplb0_mtp3_c28: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b32-eplb0-mtp3-c28.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_2p1d_dep8_b32_eplb0_mtp2_c128: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p1d-dep8-b32-eplb0-mtp2-c128.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_2p3d_dep8_b128_eplb0_mtp0_c192: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p3d-dep8-b128-eplb0-mtp0-c192.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_2p5d_tep8_b128_eplb0_mtp0_c160: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p5d-tep8-b128-eplb0-mtp0-c160.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_3p1d_dep8_b32_eplb0_mtp2_c256: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b32-eplb0-mtp2-c256.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_3p1d_dep8_b512_eplb0_mtp0_c512: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b512-eplb0-mtp0-c512.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_3p1d_dep8_b64_eplb0_mtp1_c512: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp1-c512.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_3p5d_tep8_b32_eplb0_mtp3_c160: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p5d-tep8-b32-eplb0-mtp3-c160.yaml +recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_5p3d_dep8_b256_eplb0_mtp0_c768: recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-5p3d-dep8-b256-eplb0-mtp0-c768.yaml +recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml:override_agg_b200_tp8_c1_mtp_hicache: recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c1-mtp-hicache.yaml +recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml:override_agg_b200_tp8_c4_mtp_hicache: recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c4-mtp-hicache.yaml +recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml:override_agg_b200_tp8_c8_mtp_hicache: recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c8-mtp-hicache.yaml +recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml:override_disagg_b200_1p1d_dep8_dep8_c128_mtp_kvoffload: recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-1p1d-dep8-dep8-c128-mtp-kvoffload.yaml +recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml:override_disagg_b200_1p1d_dep8_dep8_c64_mtp_kvoffload: recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml +recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml:override_disagg_b200_2p1d_dep8_dep8_c256_mtp_kvoffload: recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-2p1d-dep8-dep8-c256-mtp-kvoffload.yaml +recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp4_mtp: recipes/dsv4/sglang/gb300-fp4/agentx/agg-tp4-mtp.yaml +recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp8_mtp: recipes/dsv4/sglang/gb300-fp4/agentx/agg-tp8-mtp.yaml +recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep16_c480_mtp_kvoffload: recipes/dsv4/sglang/gb300-fp4/agentx/disagg-1p1d-dep8-dep16-c480-mtp-kvoffload.yaml +recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep16_c960_mtp_kvoffload: recipes/dsv4/sglang/gb300-fp4/agentx/disagg-2p1d-dep8-dep16-c960-mtp-kvoffload.yaml +recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_3p1d_dep8_dep16_c1440_mtp_kvoffload: recipes/dsv4/sglang/gb300-fp4/agentx/disagg-3p1d-dep8-dep16-c1440-mtp-kvoffload.yaml +recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_4p1d_dep8_dep16_c1920_mtp_kvoffload: recipes/dsv4/sglang/gb300-fp4/agentx/disagg-4p1d-dep8-dep16-c1920-mtp-kvoffload.yaml +recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep32_c388_b4_mtp: recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p1d-dep8-dep32-c388-b4-mtp.yaml +recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p4d_dep4_tep8_c4_b1_mtp: recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p4d-dep4-tep8-c4-b1-mtp.yaml +recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p6d_dep4_tep4_c24_b4_mtp: recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p6d-dep4-tep4-c24-b4-mtp.yaml +recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep32_c736_b8_mtp: recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep32-c736-b8-mtp.yaml +recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_3p1d_dep8_dep16_c1152_b32_mtp: recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-3p1d-dep8-dep16-c1152-b32-mtp.yaml +recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_5p1d_dep8_dep16_c2626_b96_mtp: recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-5p1d-dep8-dep16-c2626-b96-mtp.yaml +recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_agg_dep8_mtp: recipes/dsv4/vllm/gb200-fp4/agentx/agg-dep8-mtp.yaml +recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8_c4_mtp3: recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-c4-mtp3.yaml +recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8_c8_mtp3: recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-c8-mtp3.yaml +recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8_mtp: recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-mtp.yaml +recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep8_c128_mtp3: recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-c128-mtp3.yaml +recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep8_c256_mtp3: recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-c256-mtp3.yaml +recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep8_mtp: recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-mtp.yaml +recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep12_c576_mtp3: recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep12-c576-mtp3.yaml +recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep16_c512_mtp3: recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep16-c512-mtp3.yaml +recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep8_mtp: recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep8-mtp.yaml +recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_agg_tp4_mtp: recipes/dsv4/vllm/gb300-fp4/agentx/agg-tp4-mtp.yaml +recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_agg_tp8_mtp: recipes/dsv4/vllm/gb300-fp4/agentx/agg-tp8-mtp.yaml +recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep4_dep16_c128_mtp: recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep16-c128-mtp.yaml +recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep4_dep16_c256_mtp: recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep16-c256-mtp.yaml +recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep4_dep8_c256_mtp: recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep8-c256-mtp.yaml +recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep16_c512_mtp: recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep8-dep16-c512-mtp.yaml +recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p4d_dep4_tp8_c4_mtp: recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p4d-dep4-tp8-c4-mtp.yaml +recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p6d_dep4_tp4: recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p6d-dep4-tp4.yaml +recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep12_c1152_mtp: recipes/dsv4/vllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep12-c1152-mtp.yaml +recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep16_c1024_mtp: recipes/dsv4/vllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep16-c1024-mtp.yaml +recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_4p1d_dep4_dep8_24_c4096: recipes/dsv4/vllm/gb300-fp4/agentx/disagg-4p1d-dep4-dep8-24-c4096.yaml +recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml:override_agg_tp8_c1_mtp: recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c1-mtp.yaml +recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml:override_agg_tp8_c4_mtp: recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c4-mtp.yaml +recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml:override_agg_tp8_c8_mtp: recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c8-mtp.yaml +recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep8_c64_mtp: recipes/glm5.2/sglang/b200-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp.yaml +recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml:override_disagg_1p4d_dep8_tp4_c48_mtp: recipes/glm5.2/sglang/b200-fp4/agentx/disagg-1p4d-dep8-tp4-c48-mtp.yaml +recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp8_c2_mtp: recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c2-mtp.yaml +recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp8_c4_mtp: recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c4-mtp.yaml +recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp8_c8_mtp: recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c8-mtp.yaml +recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml:override_disagg_1p4d_dep8_tp4_c48_mtp: recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-1p4d-dep8-tp4-c48-mtp.yaml +recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml:override_disagg_1p6d_dep8_tp4_c45_mtp: recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-1p6d-dep8-tp4-c45-mtp.yaml +recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep16_c128_mtp: recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-2p1d-dep8-dep16-c128-mtp.yaml +recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tep8_c20_b5_mtp5: recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p1d-tep8-c20-b5-mtp5.yaml +recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp8_c1_b1_mtp5: recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p1d-tp8-c1-b1-mtp5.yaml +recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p4d_tep4_c30_b2_mtp5: recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p4d-tep4-c30-b2-mtp5.yaml +recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_3p4d_tep4_c60_b5_mtp5: recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-3p4d-tep4-c60-b5-mtp5.yaml +recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_4p1d_dep8_c227_b16_mtp3: recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-4p1d-dep8-c227-b16-mtp3.yaml +recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_5p1d_dep16_c260_b16_mtp3: recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-5p1d-dep16-c260-b16-mtp3.yaml +recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c1: recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c1.yaml +recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c14: recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c14.yaml +recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c24: recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c24.yaml +recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c4: recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c4.yaml +recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c48: recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c48.yaml +recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c8: recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c8.yaml +recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c96: recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c96.yaml +recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_dcp16_dspark4_maxseq2_mooncake: recipes/kimik3/vllm/gb200-fp4/agentx/agg-dcp16-dspark4-maxseq2-mooncake.yaml +recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_dcp16_nospec_mooncake: recipes/kimik3/vllm/gb200-fp4/agentx/agg-dcp16-nospec-mooncake.yaml +recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_dep16: recipes/kimik3/vllm/gb200-fp4/agentx/agg-dep16.yaml +recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_dep16_vllm_simple_offload: recipes/kimik3/vllm/gb200-fp4/agentx/agg-dep16-vllm-simple-offload.yaml +recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tep16_balanced: recipes/kimik3/vllm/gb200-fp4/agentx/agg-tep16-balanced.yaml +recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp16_latency: recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp16-latency.yaml +recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c16: recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c16.yaml +recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c32: recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c32.yaml +recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c48: recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c48.yaml +recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c72: recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c72.yaml +recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c96: recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c96.yaml +recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml:override_agg_dcp8_dspark4_mooncake: recipes/kimik3/vllm/gb300-fp4/agentx/agg-dcp8-dspark4-mooncake.yaml +recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml:override_agg_dcp8_dspark7_maxseq2_mooncake: recipes/kimik3/vllm/gb300-fp4/agentx/agg-dcp8-dspark7-maxseq2-mooncake.yaml +recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dcp8_dcp8_dspark4_mooncake: recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p1d-dcp8-dcp8-dspark4-mooncake.yaml +recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p2d_dcp8_dcp8_dspark4_mooncake: recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p2d-dcp8-dcp8-dspark4-mooncake.yaml +recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p3d_dcp8_dcp8_dspark4_mooncake: recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p3d-dcp8-dcp8-dspark4-mooncake.yaml +recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p3d_dcp8_dcp8_dspark7_mooncake: recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p3d-dcp8-dcp8-dspark7-mooncake.yaml +recipes/kimik3/vllm/h200-fp4/agentx/variants.yaml:override_agg_tp16dp2ep32_latency: recipes/kimik3/vllm/h200-fp4/agentx/agg-tp16dp2ep32-latency.yaml +recipes/kimik3/vllm/h200-fp4/agentx/variants.yaml:override_agg_tp8dp4ep32_balanced: recipes/kimik3/vllm/h200-fp4/agentx/agg-tp8dp4ep32-balanced.yaml +recipes/kimik3/vllm/h200-fp4/agentx/variants.yaml:override_agg_tp8dp4ep32_vllm_simple: recipes/kimik3/vllm/h200-fp4/agentx/agg-tp8dp4ep32-vllm-simple.yaml +recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c10_b10_eagle3: recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c10-b10-eagle3.yaml +recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c15_b15_eagle3: recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c15-b15-eagle3.yaml +recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c20_b20_eagle3: recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c20-b20-eagle3.yaml +recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c25_b25_eagle3: recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c25-b25-eagle3.yaml +recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c30_b30_eagle3: recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c30-b30-eagle3.yaml +recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c40_b40_eagle3: recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c40-b40-eagle3.yaml +recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c5_b5_eagle3: recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c5-b5-eagle3.yaml +recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_nightly_native: recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-nightly-native.yaml +recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_vllm_simple_nightly_native: recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-vllm-simple-nightly-native.yaml +recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8_nightly_native: recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp8-nightly-native.yaml +recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c24: recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-tp4-tp4-c24.yaml +recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp8_c1: recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-tp4-tp8-c1.yaml +recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_1p2d_tp4_tp4_c8_c16: recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p2d-tp4-tp4-c8-c16.yaml +recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tep4_tp4_c1: recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tep4-tp4-c1.yaml +recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tep4_tp4_c1_eval: recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tep4-tp4-c1-eval.yaml +recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp2_tp4_c20_c24: recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tp2-tp4-c20-c24.yaml +recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp2_tp4_c20_c24_eval: recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tp2-tp4-c20-c24-eval.yaml +recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p3d_dep4_tp4_c24: recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-dep4-tp4-c24.yaml +recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p3d_dep4_tp4_c24_eval: recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-dep4-tp4-c24-eval.yaml +recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p3d_tp2_tp2_c48: recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-tp2-tp2-c48.yaml +recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p3d_tp2_tp2_c48_eval: recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-tp2-tp2-c48-eval.yaml +recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p5d_tp2_tp2_c120: recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-2p5d-tp2-tp2-c120.yaml +recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p5d_tp2_tp2_c120_eval: recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-2p5d-tp2-tp2-c120-eval.yaml +recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_hicache_c16_mtp: recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c16-mtp.yaml +recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_hicache_c24_mtp: recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c24-mtp.yaml +recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_hicache_c32_mtp: recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c32-mtp.yaml +recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_hicache_c48_write_through_mtp: recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c48-write-through-mtp.yaml +recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_hicache_c64_write_through_mtp: recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c64-write-through-mtp.yaml +recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_hicache_c8_mtp: recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c8-mtp.yaml +recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c32_mtp_hicache: recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c32-mtp-hicache.yaml +recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c40_mtp_hicache: recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c40-mtp-hicache.yaml +recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c44_mtp_hicache: recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c44-mtp-hicache.yaml +recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c48_mtp_hicache: recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c48-mtp-hicache.yaml +? recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c56_replayssm_mtp_hicache +: recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c56-replayssm-mtp-hicache.yaml +recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_colocated_c12_mtp_hicache: recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c12-mtp-hicache.yaml +recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_colocated_c24_mtp_hicache: recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c24-mtp-hicache.yaml +recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_colocated_c4_mtp_hicache: recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c4-mtp-hicache.yaml +recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp4ep4_tp4_colocated_c32_mtp_hicache: recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4ep4-tp4-colocated-c32-mtp-hicache.yaml +recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp2ep2_mtp_hicache: recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache.yaml +recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp2ep2_mtp_hicache_cap48: recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache-cap48.yaml +recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp2ep2_mtp_hicache_k3_baseline: recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache-k3-baseline.yaml +recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp4_mtp: recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp.yaml +recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp4_mtp_no_symm: recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-no-symm.yaml +recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp4_mtp_parity: recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-parity.yaml +recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_b128_c1x2x8_mtp: recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c1x2x8-mtp.yaml +recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp8_ep8_d_tp8_ep8_b1024_c32x48x80_mtp: recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml +recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_1p1d_tp4_tp4_stp: recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-tp4-tp4-stp.yaml +recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_3p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c480_mtp: recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml +recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_4p1d_dep4_dep16_stp: recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-4p1d-dep4-dep16-stp.yaml +recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_4p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c768_mtp: recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml +recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_6p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c1280_mtp: recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml +recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_7p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b2048_c1344_mtp: recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml +recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_8p1d_dep4_dep16_stp: recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-8p1d-dep4-dep16-stp.yaml +? recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_8p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b2048_c1920x2304_mtp +: recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml +recipes/qwen3.5/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_b128_c1x4x8x16x32x64x256_stp: recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c1x4x8x16x32x64x256-stp.yaml +recipes/qwen3.5/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_5p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b4096_c2048_stp: recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-5p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b4096-c2048-stp.yaml +recipes/qwen3.5/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_6p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b5120_c5120_stp: recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b5120-c5120-stp.yaml +recipes/qwen3.5/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_7p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b5120_c5120_stp: recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b5120-c5120-stp.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c1_mtp_hicache_jid2530006: recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c1-mtp-hicache-jid2530006.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c24_mtp_hicache_jid2530012: recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c24-mtp-hicache-jid2530012.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c32_mtp_hicache_jid2530013: recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c32-mtp-hicache-jid2530013.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c40_mtp_hicache_jid2530015: recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c40-mtp-hicache-jid2530015.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c44_b1_mtp_hicache_nightly_c20260831: recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c44-b1-mtp-hicache-nightly-c20260831.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c44_b2_mtp_hicache_nightly_c20260831: recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c44-b2-mtp-hicache-nightly-c20260831.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c48_mtp_hicache_jid2530017: recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c48-mtp-hicache-jid2530017.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c52_mtp_hicache_jid2527406: recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c52-mtp-hicache-jid2527406.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c64_mtp_hicache_jid2527410: recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c64-mtp-hicache-jid2527410.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp8_c7_b1_mtp_hicache_nightly_c20260831: recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp8-c7-b1-mtp-hicache-nightly-c20260831.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp2_tp2_c72_mtp_hicache_session_jid2527415: recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp2-tp2-c72-mtp-hicache-session-jid2527415.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c128_mtp_hicache_session_jid2527417: recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c128-mtp-hicache-session-jid2527417.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c16_mtp_hicache_session_jid2530027: recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c16-mtp-hicache-session-jid2530027.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c32_mtp_hicache_session_jid2530028: recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c32-mtp-hicache-session-jid2530028.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c64_mtp_hicache_session_jid2530029: recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c64-mtp-hicache-session-jid2530029.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c8_mtp_hicache_session_jid2530030: recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c8-mtp-hicache-session-jid2530030.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c96_mtp_hicache_session_jid2527409: recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c96-mtp-hicache-session-jid2527409.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_3p2d_pp4_dep4_c704_mtp_hicache_nightly_c20260831: recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-3p2d-pp4-dep4-c704-mtp-hicache-nightly-c20260831.yaml +recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_3p4d_pp4_dep4_c565_mtp_hicache_nightly_c20260831: recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-3p4d-pp4-dep4-c565-mtp-hicache-nightly-c20260831.yaml +recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_b1024_c1x2x8_mtp: recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml +recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp8_ep8_d_tp8_ep8_b1024_c32x48x80_mtp: recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml +recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_1p1d_tp4_tp4_stp: recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-tp4-tp4-stp.yaml +recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_3p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c480_mtp: recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml +recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_4p1d_dep4_dep16_stp: recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-dep4-dep16-stp.yaml +recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_4p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c768_mtp: recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml +recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_6p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c1280_mtp: recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml +recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_7p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b2048_c1344_mtp: recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml +recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_8p1d_dep4_dep16_stp: recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-dep4-dep16-stp.yaml +? recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_8p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b2048_c1920x2304_mtp +: recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_10p1d_dep8_b256_eplb0_mtp0_c2150: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp0-c2150.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_11p1d_dep16_b64_eplb0_mtp0_c1076: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-11p1d-dep16-b64-eplb0-mtp0-c1076.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_11p1d_dep8_b128_eplb0_mtp3_c1229: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-11p1d-dep8-b128-eplb0-mtp3-c1229.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_16p1d_dep16_b128_eplb0_mtp0_c2253: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-16p1d-dep16-b128-eplb0-mtp0-c2253.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_17p2d_dep8_b64_eplb0_mtp3_c1126: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-17p2d-dep8-b64-eplb0-mtp3-c1126.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p2d_tep8_b16_eplb0_mtp0_c42: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p2d-tep8-b16-eplb0-mtp0-c42.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p2d_tep8_b8_eplb0_mtp3_c20: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p2d-tep8-b8-eplb0-mtp3-c20.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp0_c8: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0-c8.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp3_c12: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3-c12.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b2_eplb0_mtp3_c8: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b2-eplb0-mtp3-c8.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b4_eplb0_mtp0_c24: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp0-c24.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_24p1d_dep16_b512_eplb0_mtp0_c8192_mtp_sweep: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-24p1d-dep16-b512-eplb0-mtp0-c8192-mtp-sweep.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_24p1d_dep16_b512_eplb0_mtp0_c8192_stp_sweep: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-24p1d-dep16-b512-eplb0-mtp0-c8192-stp-sweep.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_26p1d_dep16_b256_eplb0_mtp2_c4301: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-26p1d-dep16-b256-eplb0-mtp2-c4301.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_4p1d_dep16_b16_eplb0_mtp0_c282: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p1d-dep16-b16-eplb0-mtp0-c282.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_4p3d_tep8_b32_eplb0_mtp3_c126: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p3d-tep8-b32-eplb0-mtp3-c126.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_4p3d_tep8_b64_eplb0_mtp0_c210: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p3d-tep8-b64-eplb0-mtp0-c210.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_5p1d_dep16_b8_eplb0_mtp3_c154: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-5p1d-dep16-b8-eplb0-mtp3-c154.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_7p1d_dep16_b32_eplb0_mtp0_c563: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep16-b32-eplb0-mtp0-c563.yaml +recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_7p1d_dep16_b32_eplb0_mtp3_c666: recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep16-b32-eplb0-mtp3-c666.yaml +recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep1_tep2_c44_b8_mtp_kvoffload: recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p1d-dep1-tep2-c44-b8-mtp-kvoffload.yaml +recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p7d_dep4_tep8_c7_b1_mtp_kvoffload: recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p7d-dep4-tep8-c7-b1-mtp-kvoffload.yaml +recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p2d_dep1_tep2_c52_b4_mtp_kvoffload: recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-2p2d-dep1-tep2-c52-b4-mtp-kvoffload.yaml +recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p3d_tep2_tep8_c96_b128_mtp_kvoffload: recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-2p3d-tep2-tep8-c96-b128-mtp-kvoffload.yaml +recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_3p1d_dep4_dep16_c565_b8_mtp_kvoffload: recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-3p1d-dep4-dep16-c565-b8-mtp-kvoffload.yaml +recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_3p2d_dep4_dep4_c704_b32_mtp_kvoffload: recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-3p2d-dep4-dep4-c704-b32-mtp-kvoffload.yaml diff --git a/benchmarks/multi_node/srt-slurm-recipes/RECIPES.md b/benchmarks/multi_node/srt-slurm-recipes/RECIPES.md index a731508e05..86222f7d04 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/RECIPES.md +++ b/benchmarks/multi_node/srt-slurm-recipes/RECIPES.md @@ -15,14 +15,15 @@ Store every recipe at `//-//` block per benchmark configuration, each with an explicit `name`. Master entries select one with `CONFIG_FILE=recipes//variants.yaml:override_`. Name overrides after the filename convention above, lowercase with underscores (for example `override_disagg_1p1d_dep8_b8_eplb0_mtp3`). An override lists only what differs from `base`; lists replace rather than merge, and a `null` deletes the key. Comments that apply to a single configuration sit inside its override. See [variants of one recipe](../../../docs/configuration-procedures.md#variants-of-one-recipe) for fingerprint identities. A directory keeps standalone files only where a difference cannot be expressed as an override (an explicit `null` that differs between siblings) or where deprecated configs still reference the file. +- Name other override bundles `*-variants.yaml`. Keep distinct sweep entry files separate even when their contents match: recipe paths participate in eval grouping. The Qwen3.5 `*-stp-sweep.yaml` and `*-mtp-sweep.yaml` pair preserves that existing distinction. - Update `CONFIG_FILE` and `EVAL_CONFIG_FILE` references in active and deprecated master configs, launcher path rules, workflow filters, and local documentation together when moving a file. Preserve upstream source URLs as provenance and leave historical performance-changelog entries unchanged. No aliases for the old layout are provided. Shared runtime assets stay under `configs/` beside the model directories; they are not standalone recipes. The four files in `configs/dsv4-moe-load-balancer-configs/` are copied verbatim from NVIDIA/srt-slurm commit `deb1dfd9934398664f92d194169c183e009da83b`, preserving the EPLB initial expert assignments used by 17 DSV4 TRT recipes. `setup_srt_slurm()` stages them into the job checkout's `configs/` directory for the recipes' bind mounts. Keeping a recipe in this tree does not activate it; the master configs determine the benchmark matrix. diff --git a/benchmarks/multi_node/srt-slurm-recipes/RECIPES_zh.md b/benchmarks/multi_node/srt-slurm-recipes/RECIPES_zh.md index ef3c0ed034..29d1a1ef37 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/RECIPES_zh.md +++ b/benchmarks/multi_node/srt-slurm-recipes/RECIPES_zh.md @@ -15,14 +15,15 @@ InferenceX 要求 srt-slurm 2.0 或更新版本,且配置必须声明 `schema: ```text dsr1/sglang/b200-fp4/8k1k/disagg-stp-mtp-variants.yaml glm5.2/sglang/h200-fp8/agentx/disagg-1p1d-pcp8-tp8-dp8-mtp6-hicache.yaml -qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p7d-dep4-tep8-c7-b1-mtp-kvoffload.yaml +qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml ``` - 使用主配置中的 `model-prefix` 和 `precision` 标签。引擎目录为 `sglang`、`vllm`、`trtllm` 或 `tilert`;前端仍在配置内显式声明。硬件目录使用 `b200`、`gb300` 等 GPU 型号,不使用集群名称。 - 工作负载目录为 `1k1k`、`8k1k` 或 `agentx`。已有的跨序列长度配置集合放在 `fixed-seq-len` 下,保留其覆盖项选择器。 - 文件名使用小写字母和连字符,以 `agg` 或 `disagg` 开头。包含拓扑及用于区分同目录配置的关键参数,例如并行方式、批大小、并发数、MTP、卸载或缓存设置。避免日期、带序号的延迟/吞吐量标签,以及重复目录中已有的模型或硬件信息。 - 拓扑名中的 `1p4d` 表示预填充/解码 worker 数,不一定等于物理节点数。`p-tp4` 和 `d-tp8` 分别标识预填充和解码 TP;`b` 表示批大小,`c` 表示并发数。运行参数以 YAML 为准。 -- 覆盖项集合使用 `*-variants.yaml` 命名。即使内容相同,也保留独立扫描入口:配置路径参与评估分组。Qwen3.5 的 `*-stp-sweep.yaml` 和 `*-mtp-sweep.yaml` 保留了这一既有区别。 +- 同一目录下的兄弟配置合并为一个 `variants.yaml`:共享的 `base` 加上每个基准配置一个 `override_` 块,每块显式设置 `name`。主配置通过 `CONFIG_FILE=recipes//variants.yaml:override_` 选择其一。覆盖项名称沿用上述文件命名规则,使用小写和下划线(例如 `override_disagg_1p1d_dep8_b8_eplb0_mtp3`)。覆盖项只列出与 `base` 不同的设置;列表整体替换而非合并,`null` 会删除该键。仅适用于单个配置的注释放在对应覆盖项内。指纹身份见[同一配方的多个变体](../../../docs/configuration-procedures_zh.md#同一配方的多个变体)。仅当差异无法用覆盖项表达(兄弟配置之间存在不同的显式 `null`),或已弃用配置仍引用该文件时,目录中才保留独立文件。 +- 其他覆盖项集合使用 `*-variants.yaml` 命名。即使内容相同,也保留独立扫描入口:配置路径参与评估分组。Qwen3.5 的 `*-stp-sweep.yaml` 和 `*-mtp-sweep.yaml` 保留了这一既有区别。 - 移动文件时,同步更新当前及已弃用主配置中的 `CONFIG_FILE`、`EVAL_CONFIG_FILE`,以及启动器路径规则、工作流过滤器和本地文档。保留上游来源 URL,并保持历史性能变更日志不变。不为旧目录结构提供别名。 共享运行时资源保留在模型目录旁的 `configs/` 中,不属于独立基准测试配置。`configs/dsv4-moe-load-balancer-configs/` 中的四个文件原样取自 NVIDIA/srt-slurm 提交 `deb1dfd9934398664f92d194169c183e009da83b`,保留了 17 个 DSV4 TRT 配置使用的 EPLB 初始专家分配。`setup_srt_slurm()` 将这些文件复制到作业仓库的 `configs/` 目录,供配置中的绑定挂载使用。将配置文件放入本目录不会启用该配置;实际基准测试矩阵由主配置决定。 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p1d-p-tp8-dp8-d-tp8-ep8-dp8-b256-c160x288-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p1d-p-tp8-dp8-d-tp8-ep8-dp8-b256-c160x288-mtp.yaml deleted file mode 100644 index a7f22c418a..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p1d-p-tp8-dp8-d-tp8-ep8-dp8-b256-c160x288-mtp.yaml +++ /dev/null @@ -1,148 +0,0 @@ -schema: 2 -name: b200-fp8-mtp-max-tpt-dep8-1p-2d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 256 - cuda-graph-max-bs: 32 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 -health_check: - max_attempts: 720 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: 160x288 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p1d-p-tp8-dp8-d-tp8-ep8-dp8-b256-c160x288-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p1d-p-tp8-dp8-d-tp8-ep8-dp8-b256-c160x288-stp.yaml deleted file mode 100644 index 54f90f4235..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p1d-p-tp8-dp8-d-tp8-ep8-dp8-b256-c160x288-stp.yaml +++ /dev/null @@ -1,144 +0,0 @@ -schema: 2 -name: b200-fp8-stp-max-tpt-dep8-1p-1d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 256 - cuda-graph-max-bs: 256 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 -health_check: - max_attempts: 360 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: 160x288 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p2d-p-tp8-dp8-d-tp8-ep8-dp8-b128-c288-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p2d-p-tp8-dp8-d-tp8-ep8-dp8-b128-c288-mtp.yaml deleted file mode 100644 index 55f328afa4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p2d-p-tp8-dp8-d-tp8-ep8-dp8-b128-c288-mtp.yaml +++ /dev/null @@ -1,148 +0,0 @@ -schema: 2 -name: b200-fp8-mtp-max-tpt-dep8-1p-1d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 2 - workers: 2 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 128 - cuda-graph-max-bs: 16 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 -health_check: - max_attempts: 720 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: '288' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p2d-p-tp8-dp8-d-tp8-ep8-dp8-b128-c288-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p2d-p-tp8-dp8-d-tp8-ep8-dp8-b128-c288-stp.yaml deleted file mode 100644 index 2a55518c07..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p2d-p-tp8-dp8-d-tp8-ep8-dp8-b128-c288-stp.yaml +++ /dev/null @@ -1,144 +0,0 @@ -schema: 2 -name: b200-fp8-stp-max-tpt-dep8-1p-2d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 2 - workers: 2 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 128 - cuda-graph-max-bs: 128 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 -health_check: - max_attempts: 360 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: '288' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p3d-p-tp8-dp8-d-tp8-b32-c128-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p3d-p-tp8-dp8-d-tp8-b32-c128-mtp.yaml deleted file mode 100644 index 8544eea9db..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p3d-p-tp8-dp8-d-tp8-b32-c128-mtp.yaml +++ /dev/null @@ -1,145 +0,0 @@ -schema: 2 -name: b200-fp8-mtp-low-latency-tep8-1p-1d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 3 - workers: 3 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 32 - cuda-graph-max-bs: 32 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 1 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 -health_check: - max_attempts: 720 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: '128' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p3d-p-tp8-dp8-d-tp8-b32-c128-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p3d-p-tp8-dp8-d-tp8-b32-c128-stp.yaml deleted file mode 100644 index 32d3bdef06..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p3d-p-tp8-dp8-d-tp8-b32-c128-stp.yaml +++ /dev/null @@ -1,143 +0,0 @@ -schema: 2 -name: b200-fp8-stp-low-latency-tp8-1p-3d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 3 - workers: 3 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 32 - cuda-graph-max-bs: 32 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 1 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - # disable-chunked-prefix-cache: true - -health_check: - max_attempts: 360 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: '128' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p4d-p-tp8-dp8-d-tp8-b32-c128-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p4d-p-tp8-dp8-d-tp8-b32-c128-mtp.yaml deleted file mode 100644 index d8cda49d82..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p4d-p-tp8-dp8-d-tp8-b32-c128-mtp.yaml +++ /dev/null @@ -1,145 +0,0 @@ -schema: 2 -name: b200-fp8-mtp-low-latency-tep8-1p-4d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 4 - workers: 4 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 32 - cuda-graph-max-bs: 32 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 1 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 -health_check: - max_attempts: 720 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: '128' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p4d-p-tp8-dp8-d-tp8-b32-c128-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p4d-p-tp8-dp8-d-tp8-b32-c128-stp.yaml deleted file mode 100644 index b4f73851b9..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p4d-p-tp8-dp8-d-tp8-b32-c128-stp.yaml +++ /dev/null @@ -1,143 +0,0 @@ -schema: 2 -name: b200-fp8-stp-low-latency-tp8-1p-4d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 4 - workers: 4 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 32 - cuda-graph-max-bs: 32 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 1 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - # disable-chunked-prefix-cache: true - -health_check: - max_attempts: 360 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: '128' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p6d-p-tp8-dp8-d-tp8-b22-c8x16x32x64x128-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p6d-p-tp8-dp8-d-tp8-b22-c8x16x32x64x128-mtp.yaml deleted file mode 100644 index 320d9de2a5..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p6d-p-tp8-dp8-d-tp8-b22-c8x16x32x64x128-mtp.yaml +++ /dev/null @@ -1,145 +0,0 @@ -schema: 2 -name: b200-fp8-mtp-low-latency-tep8-1p-6d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 6 - workers: 6 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 22 - cuda-graph-max-bs: 22 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 1 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 -health_check: - max_attempts: 720 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: 8x16x32x64x128 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p6d-p-tp8-dp8-d-tp8-b22-c8x16x32x64x128-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p6d-p-tp8-dp8-d-tp8-b22-c8x16x32x64x128-stp.yaml deleted file mode 100644 index c045d20de4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p6d-p-tp8-dp8-d-tp8-b22-c8x16x32x64x128-stp.yaml +++ /dev/null @@ -1,143 +0,0 @@ -schema: 2 -name: b200-fp8-stp-low-latency-tp8-1p-6d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 6 - workers: 6 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 22 - cuda-graph-max-bs: 22 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 1 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - # disable-chunked-prefix-cache: true - -health_check: - max_attempts: 360 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: 8x16x32x64x128 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-2p1d-p-tp8-dp8-d-tp8-ep8-dp8-b512-c512-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-2p1d-p-tp8-dp8-d-tp8-ep8-dp8-b512-c512-mtp.yaml deleted file mode 100644 index 7c96ef3bce..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-2p1d-p-tp8-dp8-d-tp8-ep8-dp8-b512-c512-mtp.yaml +++ /dev/null @@ -1,148 +0,0 @@ -schema: 2 -name: b200-fp8-mtp-max-tpt-dep8-2p-1d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 2 - workers: 2 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 512 - cuda-graph-max-bs: 64 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 -health_check: - max_attempts: 720 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: '512' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-2p1d-p-tp8-dp8-d-tp8-ep8-dp8-b512-c512-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-2p1d-p-tp8-dp8-d-tp8-ep8-dp8-b512-c512-stp.yaml deleted file mode 100644 index b26a35931f..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-2p1d-p-tp8-dp8-d-tp8-ep8-dp8-b512-c512-stp.yaml +++ /dev/null @@ -1,144 +0,0 @@ -schema: 2 -name: b200-fp8-stp-max-tpt-dep8-2p-1d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 2 - workers: 2 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 512 - cuda-graph-max-bs: 512 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 -health_check: - max_attempts: 360 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: '512' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-3p1d-p-tp8-dp8-d-tp8-ep8-dp8-b1024-c1024-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-3p1d-p-tp8-dp8-d-tp8-ep8-dp8-b1024-c1024-mtp.yaml deleted file mode 100644 index d1d1a67f5d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-3p1d-p-tp8-dp8-d-tp8-ep8-dp8-b1024-c1024-mtp.yaml +++ /dev/null @@ -1,148 +0,0 @@ -schema: 2 -name: b200-fp8-mtp-max-tpt-dep8-3p-1d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 3 - workers: 3 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - SGLANG_ENABLE_SPEC_V2: '1' - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 1024 - cuda-graph-max-bs: 128 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 -health_check: - max_attempts: 720 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: '1024' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-3p1d-p-tp8-dp8-d-tp8-ep8-dp8-b1024-c1024-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-3p1d-p-tp8-dp8-d-tp8-ep8-dp8-b1024-c1024-stp.yaml deleted file mode 100644 index 8f6cc2201c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/disagg-3p1d-p-tp8-dp8-d-tp8-ep8-dp8-b1024-c1024-stp.yaml +++ /dev/null @@ -1,144 +0,0 @@ -schema: 2 -name: b200-fp8-stp-max-tpt-dep8-3p-1d - -dynamo: - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - source: - pypi: 0.9.1 - -model: - path: dsr1-fp8 - container: dynamo-sglang - precision: fp8 - -resources: - gpu_type: b200 - gpus_per_node: 8 - -engine: sglang -roles: - prefill: - nodes: 3 - workers: 3 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - load-balance-method: round_robin - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 8192 - chunked-prefill-size: 65536 - max-running-requests: 8 - context-length: 9600 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - moe-dense-tp-size: 1 - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - - decode: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - CUDA_SCALE_LAUNCH_QUEUES: 4x - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' - - args: - # Model configuration - served-model-name: deepseek-ai/DeepSeek-R1 - trust-remote-code: true - quantization: fp8 - - # Disaggregation mode - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - context-length: 9600 - max-running-requests: 1024 - cuda-graph-max-bs: 1024 - - # Parallelism - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - - # Attention - attention-backend: trtllm_mla - kv-cache-dtype: fp8_e4m3 - - # MoE - moe-runner-backend: flashinfer_trtllm - - # Other flags - stream-interval: 30 - watchdog-timeout: 1000000 - enable-flashinfer-allreduce-fusion: true - disable-radix-cache: true - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 -health_check: - max_attempts: 360 - interval_seconds: 10 - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - req_rate: inf - concurrencies: '1024' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml new file mode 100644 index 0000000000..af5a46d980 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml @@ -0,0 +1,465 @@ +# srt-slurm recipes for dsr1/sglang/b200-fp8/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_. + +schema: 2 + +base: + dynamo: + # Use NATS for a recipe prior to Dynamo commit 39d2a68. + request_plane: nats + source: + pypi: 0.9.1 + model: + path: dsr1-fp8 + container: dynamo-sglang + precision: fp8 + resources: + gpu_type: b200 + gpus_per_node: 8 + engine: sglang + roles: + prefill: + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + CUDA_SCALE_LAUNCH_QUEUES: 4x + SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' + args: + # Model configuration + served-model-name: deepseek-ai/DeepSeek-R1 + trust-remote-code: true + quantization: fp8 + # Disaggregation mode + disaggregation-mode: prefill + disaggregation-transfer-backend: nixl + load-balance-method: round_robin + # Memory and token limits + mem-fraction-static: 0.6 + max-prefill-tokens: 8192 + chunked-prefill-size: 65536 + max-running-requests: 8 + context-length: 9600 + # Parallelism + tensor-parallel-size: 8 + data-parallel-size: 8 + expert-parallel-size: 1 + enable-dp-attention: true + enable-dp-lm-head: true + # Attention + attention-backend: trtllm_mla + kv-cache-dtype: fp8_e4m3 + # MoE + moe-runner-backend: flashinfer_trtllm + moe-dense-tp-size: 1 + # Other flags + stream-interval: 30 + watchdog-timeout: 1000000 + enable-flashinfer-allreduce-fusion: true + disable-radix-cache: true + decode: + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + CUDA_SCALE_LAUNCH_QUEUES: 4x + SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' + args: + # Model configuration + served-model-name: deepseek-ai/DeepSeek-R1 + trust-remote-code: true + quantization: fp8 + # Disaggregation mode + disaggregation-mode: decode + disaggregation-transfer-backend: nixl + # Memory and token limits + mem-fraction-static: 0.75 + context-length: 9600 + # Parallelism + tensor-parallel-size: 8 + # Attention + attention-backend: trtllm_mla + kv-cache-dtype: fp8_e4m3 + # MoE + moe-runner-backend: flashinfer_trtllm + # Other flags + stream-interval: 30 + watchdog-timeout: 1000000 + enable-flashinfer-allreduce-fusion: true + disable-radix-cache: true + health_check: + interval_seconds: 10 + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + +override_disagg_1p1d_p_tp8_dp8_d_tp8_ep8_dp8_b256_c160x288_mtp: + name: b200-fp8-mtp-max-tpt-dep8-1p-2d + roles: + prefill: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + decode: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + max-running-requests: 256 + cuda-graph-max-bs: 32 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + health_check: + max_attempts: 720 + benchmark: + concurrencies: 160x288 + +override_disagg_1p1d_p_tp8_dp8_d_tp8_ep8_dp8_b256_c160x288_stp: + name: b200-fp8-stp-max-tpt-dep8-1p-1d + roles: + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 1 + workers: 1 + args: + max-running-requests: 256 + cuda-graph-max-bs: 256 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + health_check: + max_attempts: 360 + benchmark: + concurrencies: 160x288 + +override_disagg_1p2d_p_tp8_dp8_d_tp8_ep8_dp8_b128_c288_mtp: + name: b200-fp8-mtp-max-tpt-dep8-1p-1d + roles: + prefill: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + decode: + nodes: 2 + workers: 2 + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + max-running-requests: 128 + cuda-graph-max-bs: 16 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + health_check: + max_attempts: 720 + benchmark: + concurrencies: '288' + +override_disagg_1p2d_p_tp8_dp8_d_tp8_ep8_dp8_b128_c288_stp: + name: b200-fp8-stp-max-tpt-dep8-1p-2d + roles: + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 2 + workers: 2 + args: + max-running-requests: 128 + cuda-graph-max-bs: 128 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + health_check: + max_attempts: 360 + benchmark: + concurrencies: '288' + +override_disagg_1p3d_p_tp8_dp8_d_tp8_b32_c128_mtp: + name: b200-fp8-mtp-low-latency-tep8-1p-1d + roles: + prefill: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + decode: + nodes: 3 + workers: 3 + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + max-running-requests: 32 + cuda-graph-max-bs: 32 + data-parallel-size: 1 + expert-parallel-size: 1 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + health_check: + max_attempts: 720 + benchmark: + concurrencies: '128' + +override_disagg_1p3d_p_tp8_dp8_d_tp8_b32_c128_stp: + name: b200-fp8-stp-low-latency-tp8-1p-3d + roles: + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 3 + workers: 3 + args: + max-running-requests: 32 + cuda-graph-max-bs: 32 + data-parallel-size: 1 + expert-parallel-size: 1 + # disable-chunked-prefix-cache: true + health_check: + max_attempts: 360 + benchmark: + concurrencies: '128' + +override_disagg_1p4d_p_tp8_dp8_d_tp8_b32_c128_mtp: + name: b200-fp8-mtp-low-latency-tep8-1p-4d + roles: + prefill: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + decode: + nodes: 4 + workers: 4 + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + max-running-requests: 32 + cuda-graph-max-bs: 32 + data-parallel-size: 1 + expert-parallel-size: 1 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + health_check: + max_attempts: 720 + benchmark: + concurrencies: '128' + +override_disagg_1p4d_p_tp8_dp8_d_tp8_b32_c128_stp: + name: b200-fp8-stp-low-latency-tp8-1p-4d + roles: + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 4 + workers: 4 + args: + max-running-requests: 32 + cuda-graph-max-bs: 32 + data-parallel-size: 1 + expert-parallel-size: 1 + # disable-chunked-prefix-cache: true + health_check: + max_attempts: 360 + benchmark: + concurrencies: '128' + +override_disagg_1p6d_p_tp8_dp8_d_tp8_b22_c8x16x32x64x128_mtp: + name: b200-fp8-mtp-low-latency-tep8-1p-6d + roles: + prefill: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + decode: + nodes: 6 + workers: 6 + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + max-running-requests: 22 + cuda-graph-max-bs: 22 + data-parallel-size: 1 + expert-parallel-size: 1 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + health_check: + max_attempts: 720 + benchmark: + concurrencies: 8x16x32x64x128 + +override_disagg_1p6d_p_tp8_dp8_d_tp8_b22_c8x16x32x64x128_stp: + name: b200-fp8-stp-low-latency-tp8-1p-6d + roles: + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 6 + workers: 6 + args: + max-running-requests: 22 + cuda-graph-max-bs: 22 + data-parallel-size: 1 + expert-parallel-size: 1 + # disable-chunked-prefix-cache: true + health_check: + max_attempts: 360 + benchmark: + concurrencies: 8x16x32x64x128 + +override_disagg_2p1d_p_tp8_dp8_d_tp8_ep8_dp8_b512_c512_mtp: + name: b200-fp8-mtp-max-tpt-dep8-2p-1d + roles: + prefill: + nodes: 2 + workers: 2 + env: + SGLANG_ENABLE_SPEC_V2: '1' + decode: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + max-running-requests: 512 + cuda-graph-max-bs: 64 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + health_check: + max_attempts: 720 + benchmark: + concurrencies: '512' + +override_disagg_2p1d_p_tp8_dp8_d_tp8_ep8_dp8_b512_c512_stp: + name: b200-fp8-stp-max-tpt-dep8-2p-1d + roles: + prefill: + nodes: 2 + workers: 2 + decode: + nodes: 1 + workers: 1 + args: + max-running-requests: 512 + cuda-graph-max-bs: 512 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + health_check: + max_attempts: 360 + benchmark: + concurrencies: '512' + +override_disagg_3p1d_p_tp8_dp8_d_tp8_ep8_dp8_b1024_c1024_mtp: + name: b200-fp8-mtp-max-tpt-dep8-3p-1d + roles: + prefill: + nodes: 3 + workers: 3 + env: + SGLANG_ENABLE_SPEC_V2: '1' + decode: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + max-running-requests: 1024 + cuda-graph-max-bs: 128 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + health_check: + max_attempts: 720 + benchmark: + concurrencies: '1024' + +override_disagg_3p1d_p_tp8_dp8_d_tp8_ep8_dp8_b1024_c1024_stp: + name: b200-fp8-stp-max-tpt-dep8-3p-1d + roles: + prefill: + nodes: 3 + workers: 3 + decode: + nodes: 1 + workers: 1 + args: + max-running-requests: 1024 + cuda-graph-max-bs: 1024 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + health_check: + max_attempts: 360 + benchmark: + concurrencies: '1024' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-10p1d-p-tp4-d-tp32-ep32-dp32-b16384-c2048-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-10p1d-p-tp4-d-tp32-ep32-dp32-b16384-c2048-stp.yaml deleted file mode 100644 index 6d5f76a41e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-10p1d-p-tp4-d-tp32-ep32-dp32-b16384-c2048-stp.yaml +++ /dev/null @@ -1,183 +0,0 @@ -schema: 2 -name: "gb200-fp4-8k1k-max-tpt" - -dynamo: - request_plane: "nats" - - source: - pypi: 0.8.1 - # Use NATS for a recipe prior to Dynamo commit 39d2a68. -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 9 - nginx_container: nginx-sqsh - -model: - path: "dsr1-fp4" - container: "dynamo-sglang" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: sglang -roles: - prefill: - nodes: 10 - workers: 10 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: "1" - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - MC_TE_METRIC: "true" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - trust-remote-code: true - - # KV cache and attention - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - - # Quantization - quantization: "modelopt_fp4" - moe-runner-backend: "flashinfer_trtllm" - - # Radix cache disabled - disable-radix-cache: true - disable-chunked-prefix-cache: true - - # Other flags - stream-interval: 50 - decode-log-interval: 1000 - watchdog-timeout: 1000000 - context-length: 9600 - disable-shared-experts-fusion: true - disaggregation-bootstrap-port: 30001 - - # Prefill-specific mode - disaggregation-mode: "prefill" - - # Memory and token limits - mem-fraction-static: 0.95 - max-total-tokens: 131072 - max-prefill-tokens: 524288 - chunked-prefill-size: 131072 - - # Request handling - max-running-requests: 30000 - load-balance-method: "round_robin" - - # Performance optimizations - disable-cuda-graph: true - enable-dp-attention: false - fp4-gemm-backend: "flashinfer_cutlass" - disaggregation-transfer-backend: nixl - - # Parallelism - tp-size: 4 - dp-size: 1 - ep-size: 1 - - decode: - nodes: 8 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: "1" - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - MC_TE_METRIC: "true" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_MOE_NVFP4_DISPATCH: "1" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - trust-remote-code: true - - # KV cache and attention - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - - # Quantization - quantization: "modelopt_fp4" - moe-runner-backend: "flashinfer_cutedsl" - - # Radix cache disabled - disable-radix-cache: true - disable-chunked-prefix-cache: true - - # Other flags - stream-interval: 50 - decode-log-interval: 1000 - watchdog-timeout: 1000000 - context-length: 9600 - disable-shared-experts-fusion: true - eplb-algorithm: "deepseek" - disaggregation-bootstrap-port: 30001 - - # Decode-specific mode - disaggregation-mode: "decode" - - # Memory and token limits - mem-fraction-static: 0.83 - max-total-tokens: 524288 - chunked-prefill-size: 24576 - - # Request handling - max-running-requests: 16384 - - # DeepEP configuration - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - ep-num-redundant-experts: 32 - - cuda-graph-max-bs: 512 - num-reserved-decode-tokens: 112 - - # Additional decode optimizations - moe-dense-tp-size: 1 - enable-dp-lm-head: true - prefill-round-robin-balance: true - enable-dp-attention: true - fp4-gemm-backend: "flashinfer_cutlass" - disaggregation-transfer-backend: nixl - - # Parallelism - tp-size: 32 - dp-size: 32 - ep-size: 32 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "2048" - req_rate: 700 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-1p4d-p-tp4-d-tp4-c4x8-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-1p4d-p-tp4-d-tp4-c4x8-stp.yaml deleted file mode 100644 index 0793b450a9..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-1p4d-p-tp4-d-tp4-c4x8-stp.yaml +++ /dev/null @@ -1,123 +0,0 @@ -schema: 2 -name: "gb200-fp4-8k1k-low-latency" - -dynamo: - request_plane: "nats" - - source: - pypi: 0.8.1 - # Use NATS for a recipe prior to Dynamo commit 39d2a68. -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 4 - nginx_container: nginx-sqsh - -model: - path: "dsr1-fp4" - container: "dynamo-sglang" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_ENABLE_JIT_DEEPGEMM: "false" - - args: - disaggregation-mode: "prefill" - served-model-name: "deepseek-ai/DeepSeek-R1" - trust-remote-code: true - disable-radix-cache: true - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - quantization: "modelopt_fp4" - moe-runner-backend: "flashinfer_trtllm" - stream-interval: 50 - watchdog-timeout: 1000000 - context-length: 9600 - mem-fraction-static: 0.95 - max-total-tokens: 32768 - chunked-prefill-size: 24576 - cuda-graph-max-bs: 256 - max-running-requests: 512 - scheduler-recv-interval: 10 - enable-symm-mem: true - load-balance-method: "round_robin" - disaggregation-bootstrap-port: 30001 - data-parallel-size: 1 - disaggregation-transfer-backend: nixl - fp4-gemm-backend: "flashinfer_trtllm" - tensor-parallel-size: 4 - expert-parallel-size: 1 - enable-dp-attention: false - - decode: - nodes: 4 - workers: 4 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_ENABLE_JIT_DEEPGEMM: "false" - - args: - disaggregation-mode: "decode" - served-model-name: "deepseek-ai/DeepSeek-R1" - prefill-round-robin-balance: true - trust-remote-code: true - disable-radix-cache: true - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - quantization: "modelopt_fp4" - moe-runner-backend: "flashinfer_trtllm" - disaggregation-bootstrap-port: 30001 - stream-interval: 50 - watchdog-timeout: 1000000 - context-length: 9600 - mem-fraction-static: 0.95 - chunked-prefill-size: 8192 - cuda-graph-max-bs: 256 - scheduler-recv-interval: 10 - enable-symm-mem: true - disaggregation-transfer-backend: nixl - fp4-gemm-backend: "flashinfer_trtllm" - tensor-parallel-size: 4 - expert-parallel-size: 1 - enable-dp-attention: false - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "4x8" - req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-6p1d-p-tp4-d-tp48-ep48-dp48-b16384-c512x2048x4096-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-6p1d-p-tp4-d-tp48-ep48-dp48-b16384-c512x2048x4096-stp.yaml deleted file mode 100644 index db1ef476f2..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-6p1d-p-tp4-d-tp48-ep48-dp48-b16384-c512x2048x4096-stp.yaml +++ /dev/null @@ -1,183 +0,0 @@ -schema: 2 -name: "gb200-fp4-8k1k-mid-curve" - -dynamo: - request_plane: "nats" - - source: - pypi: 0.8.1 - # Use NATS for a recipe prior to Dynamo commit 39d2a68. -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 9 - nginx_container: nginx-sqsh - -model: - path: "dsr1-fp4" - container: "dynamo-sglang" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: sglang -roles: - prefill: - nodes: 6 - workers: 6 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: "1" - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - MC_TE_METRIC: "true" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - trust-remote-code: true - - # KV cache and attention - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - - # Quantization - quantization: "modelopt_fp4" - moe-runner-backend: "flashinfer_trtllm" - - # Radix cache disabled - disable-radix-cache: true - disable-chunked-prefix-cache: true - - # Other flags - stream-interval: 50 - decode-log-interval: 1000 - watchdog-timeout: 1000000 - context-length: 9600 - disable-shared-experts-fusion: true - disaggregation-bootstrap-port: 30001 - - # Prefill-specific mode - disaggregation-mode: "prefill" - - # Memory and token limits - mem-fraction-static: 0.95 - max-total-tokens: 131072 - max-prefill-tokens: 524288 - chunked-prefill-size: 131072 - - # Request handling - max-running-requests: 30000 - load-balance-method: "round_robin" - - # Performance optimizations - disable-cuda-graph: true - enable-dp-attention: false - fp4-gemm-backend: "flashinfer_cutlass" - disaggregation-transfer-backend: nixl - - # Parallelism - tp-size: 4 - dp-size: 1 - ep-size: 1 - - decode: - nodes: 12 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: "1" - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - MC_TE_METRIC: "true" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_MOE_NVFP4_DISPATCH: "1" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - trust-remote-code: true - - # KV cache and attention - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - - # Quantization - quantization: "modelopt_fp4" - moe-runner-backend: "flashinfer_cutedsl" - - # Radix cache disabled - disable-radix-cache: true - disable-chunked-prefix-cache: true - - # Other flags - stream-interval: 50 - decode-log-interval: 1000 - watchdog-timeout: 1000000 - context-length: 9600 - disable-shared-experts-fusion: true - eplb-algorithm: "deepseek" - disaggregation-bootstrap-port: 30001 - - # Decode-specific mode - disaggregation-mode: "decode" - - # Memory and token limits - mem-fraction-static: 0.83 - max-total-tokens: 524288 - chunked-prefill-size: 24576 - - # Request handling - max-running-requests: 16384 - - # DeepEP configuration - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - ep-num-redundant-experts: 32 - - cuda-graph-max-bs: 512 - num-reserved-decode-tokens: 112 - - # Additional decode optimizations - moe-dense-tp-size: 1 - enable-dp-lm-head: true - prefill-round-robin-balance: true - enable-dp-attention: true - fp4-gemm-backend: "flashinfer_cutlass" - disaggregation-transfer-backend: nixl - - # Parallelism - tp-size: 48 - dp-size: 48 - ep-size: 48 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "512x2048x4096" - req_rate: 700 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp4/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp4/8k1k/variants.yaml new file mode 100644 index 0000000000..2b93da7df1 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp4/8k1k/variants.yaml @@ -0,0 +1,297 @@ +# srt-slurm recipes for dsr1/sglang/gb200-fp4/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/sglang/gb200-fp4/8k1k/variants.yaml:override_. + +schema: 2 + +base: + dynamo: + request_plane: nats + source: + pypi: 0.8.1 + # Use NATS for a recipe prior to Dynamo commit 39d2a68. + frontend: + type: dynamo + enable_multiple_frontends: true + nginx_container: nginx-sqsh + model: + path: dsr1-fp4 + container: dynamo-sglang + precision: fp4 + resources: + gpu_type: gb200 + gpus_per_node: 4 + engine: sglang + roles: + prefill: + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + args: + served-model-name: deepseek-ai/DeepSeek-R1 + trust-remote-code: true + kv-cache-dtype: fp8_e4m3 + attention-backend: trtllm_mla + quantization: modelopt_fp4 + moe-runner-backend: flashinfer_trtllm + disable-radix-cache: true + stream-interval: 50 + watchdog-timeout: 1000000 + context-length: 9600 + disaggregation-bootstrap-port: 30001 + disaggregation-mode: prefill + mem-fraction-static: 0.95 + load-balance-method: round_robin + enable-dp-attention: false + disaggregation-transfer-backend: nixl + decode: + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + args: + served-model-name: deepseek-ai/DeepSeek-R1 + trust-remote-code: true + kv-cache-dtype: fp8_e4m3 + attention-backend: trtllm_mla + quantization: modelopt_fp4 + disable-radix-cache: true + stream-interval: 50 + watchdog-timeout: 1000000 + context-length: 9600 + disaggregation-bootstrap-port: 30001 + disaggregation-mode: decode + prefill-round-robin-balance: true + disaggregation-transfer-backend: nixl + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + +override_disagg_10p1d_p_tp4_d_tp32_ep32_dp32_b16384_c2048_stp: + name: gb200-fp4-8k1k-max-tpt + frontend: + num_additional_frontends: 9 + roles: + prefill: + nodes: 10 + workers: 10 + env: + SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: '1' + SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' + MC_TE_METRIC: 'true' + # Decode-specific environment variables + # Model configuration + # KV cache and attention + # Quantization + # Radix cache disabled + # Other flags + # Prefill-specific mode + # Memory and token limits + args: + disable-chunked-prefix-cache: true + decode-log-interval: 1000 + disable-shared-experts-fusion: true + max-total-tokens: 131072 + max-prefill-tokens: 524288 + chunked-prefill-size: 131072 + # Request handling + max-running-requests: 30000 + # Performance optimizations + disable-cuda-graph: true + fp4-gemm-backend: flashinfer_cutlass + # Parallelism + tp-size: 4 + dp-size: 1 + ep-size: 1 + decode: + nodes: 8 + workers: 1 + env: + SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: '1' + SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' + MC_TE_METRIC: 'true' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + SGLANG_MOE_NVFP4_DISPATCH: '1' + # Model configuration + # KV cache and attention + # Quantization + # Radix cache disabled + # Other flags + # Decode-specific mode + args: + moe-runner-backend: flashinfer_cutedsl + disable-chunked-prefix-cache: true + decode-log-interval: 1000 + disable-shared-experts-fusion: true + eplb-algorithm: deepseek + # Memory and token limits + mem-fraction-static: 0.83 + max-total-tokens: 524288 + chunked-prefill-size: 24576 + # Request handling + max-running-requests: 16384 + # DeepEP configuration + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + ep-num-redundant-experts: 32 + cuda-graph-max-bs: 512 + num-reserved-decode-tokens: 112 + # Additional decode optimizations + moe-dense-tp-size: 1 + enable-dp-lm-head: true + enable-dp-attention: true + fp4-gemm-backend: flashinfer_cutlass + # Parallelism + tp-size: 32 + dp-size: 32 + ep-size: 32 + benchmark: + concurrencies: '2048' + req_rate: 700 + +override_disagg_1p4d_p_tp4_d_tp4_c4x8_stp: + name: gb200-fp4-8k1k-low-latency + frontend: + num_additional_frontends: 4 + roles: + prefill: + nodes: 1 + workers: 1 + env: + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_ENABLE_JIT_DEEPGEMM: 'false' + args: + max-total-tokens: 32768 + chunked-prefill-size: 24576 + max-running-requests: 512 + fp4-gemm-backend: flashinfer_trtllm + cuda-graph-max-bs: 256 + scheduler-recv-interval: 10 + enable-symm-mem: true + data-parallel-size: 1 + tensor-parallel-size: 4 + expert-parallel-size: 1 + decode: + nodes: 4 + workers: 4 + env: + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_ENABLE_JIT_DEEPGEMM: 'false' + args: + moe-runner-backend: flashinfer_trtllm + mem-fraction-static: 0.95 + chunked-prefill-size: 8192 + cuda-graph-max-bs: 256 + enable-dp-attention: false + fp4-gemm-backend: flashinfer_trtllm + scheduler-recv-interval: 10 + enable-symm-mem: true + tensor-parallel-size: 4 + expert-parallel-size: 1 + benchmark: + concurrencies: 4x8 + req_rate: 300 + +override_disagg_6p1d_p_tp4_d_tp48_ep48_dp48_b16384_c512x2048x4096_stp: + name: gb200-fp4-8k1k-mid-curve + frontend: + num_additional_frontends: 9 + roles: + prefill: + nodes: 6 + workers: 6 + env: + SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: '1' + SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' + MC_TE_METRIC: 'true' + # Decode-specific environment variables + # Model configuration + # KV cache and attention + # Quantization + # Radix cache disabled + # Other flags + # Prefill-specific mode + # Memory and token limits + args: + disable-chunked-prefix-cache: true + decode-log-interval: 1000 + disable-shared-experts-fusion: true + max-total-tokens: 131072 + max-prefill-tokens: 524288 + chunked-prefill-size: 131072 + # Request handling + max-running-requests: 30000 + # Performance optimizations + disable-cuda-graph: true + fp4-gemm-backend: flashinfer_cutlass + # Parallelism + tp-size: 4 + dp-size: 1 + ep-size: 1 + decode: + nodes: 12 + workers: 1 + env: + SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: '1' + SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' + MC_TE_METRIC: 'true' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + SGLANG_MOE_NVFP4_DISPATCH: '1' + # Model configuration + # KV cache and attention + # Quantization + # Radix cache disabled + # Other flags + # Decode-specific mode + args: + moe-runner-backend: flashinfer_cutedsl + disable-chunked-prefix-cache: true + decode-log-interval: 1000 + disable-shared-experts-fusion: true + eplb-algorithm: deepseek + # Memory and token limits + mem-fraction-static: 0.83 + max-total-tokens: 524288 + chunked-prefill-size: 24576 + # Request handling + max-running-requests: 16384 + # DeepEP configuration + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + ep-num-redundant-experts: 32 + cuda-graph-max-bs: 512 + num-reserved-decode-tokens: 112 + # Additional decode optimizations + moe-dense-tp-size: 1 + enable-dp-lm-head: true + enable-dp-attention: true + fp4-gemm-backend: flashinfer_cutlass + # Parallelism + tp-size: 48 + dp-size: 48 + ep-size: 48 + benchmark: + concurrencies: 512x2048x4096 + req_rate: 700 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp8/8k1k/disagg-5p1d-p-tp8-ep8-dp8-d-tp32-ep32-dp32-b8192-c512x1024x2048x6144-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp8/8k1k/disagg-5p1d-p-tp8-ep8-dp8-d-tp32-ep32-dp32-b8192-c512x1024x2048x6144-stp.yaml deleted file mode 100644 index 344a5b3391..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp8/8k1k/disagg-5p1d-p-tp8-ep8-dp8-d-tp32-ep32-dp32-b8192-c512x1024x2048x6144-stp.yaml +++ /dev/null @@ -1,174 +0,0 @@ -schema: 2 -name: "gb200-8k1k-fp8-mid-tpt" - -dynamo: - request_plane: "nats" - - source: - pypi: 0.8.1 - # Use NATS for a recipe prior to Dynamo commit 39d2a68. -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 9 - nginx_container: nginx - -model: - path: "dsr1-fp8" - container: "dynamo-sglang" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: sglang -roles: - prefill: - nodes: 10 - workers: 5 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - MC_TE_METRIC: "true" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - PYTHONUNBUFFERED: "1" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - - # KV cache and attention - attention-backend: "trtllm_mla" - kv-cache-dtype: "fp8_e4m3" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - stream-interval: 50 - max-running-requests: 30000 - context-length: 9300 - watchdog-timeout: 1000000 - disable-shared-experts-fusion: true - eplb-algorithm: "deepseek" - - # Prefill-specific mode - disaggregation-mode: "prefill" - - # Memory and token limits - mem-fraction-static: 0.80 - max-total-tokens: 524288 - chunked-prefill-size: 131072 - - # Request handling - load-balance-method: "round_robin" - - # Performance optimizations - disable-cuda-graph: true - - # DeepEP configuration - moe-a2a-backend: "deepep" - deepep-mode: "normal" - ep-dispatch-algorithm: "dynamic" - moe-dense-tp-size: 1 - enable-dp-lm-head: true - ep-num-redundant-experts: 32 - deepep-config: "/configs/deepep_config.json" - disaggregation-bootstrap-port: 30001 - disaggregation-transfer-backend: nixl - - decode: - nodes: 8 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "256" - MC_TE_METRIC: "true" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - PYTHONUNBUFFERED: "1" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 32 - dp-size: 32 - ep-size: 32 - enable-dp-attention: true - - # KV cache and attention - attention-backend: "trtllm_mla" - kv-cache-dtype: "fp8_e4m3" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - stream-interval: 50 - decode-log-interval: 1000 - max-running-requests: 8192 - context-length: 9300 - watchdog-timeout: 1000000 - disable-shared-experts-fusion: true - eplb-algorithm: "deepseek" - - # Decode-specific mode - disaggregation-mode: "decode" - - # Memory and token limits - mem-fraction-static: 0.82 - chunked-prefill-size: 36864 - - # DeepEP configuration - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - moe-dense-tp-size: 1 - enable-dp-lm-head: true - prefill-round-robin-balance: true - ep-num-redundant-experts: 32 - deepep-config: "/configs/deepep_config.json" - # CUDA graphs - cuda-graph-max-bs: 256 - disaggregation-bootstrap-port: 30001 - disaggregation-transfer-backend: nixl - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "512x1024x2048x6144" - req_rate: "300" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp8/8k1k/disagg-6p1d-p-tp8-ep8-dp8-d-tp24-ep24-dp24-b8192-c2048x4096x6144-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp8/8k1k/disagg-6p1d-p-tp8-ep8-dp8-d-tp24-ep24-dp24-b8192-c2048x4096x6144-stp.yaml deleted file mode 100644 index 6a22f63b68..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp8/8k1k/disagg-6p1d-p-tp8-ep8-dp8-d-tp24-ep24-dp24-b8192-c2048x4096x6144-stp.yaml +++ /dev/null @@ -1,177 +0,0 @@ -schema: 2 -name: "gb200-8k1k-fp8-max-tpt" - -dynamo: - request_plane: "nats" - - source: - pypi: 0.8.1 - # Use NATS for a recipe prior to Dynamo commit 39d2a68. -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 9 - nginx_container: nginx - -model: - path: "dsr1-fp8" - container: "dynamo-sglang" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: sglang -roles: - prefill: - nodes: 12 - workers: 6 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - MC_TE_METRIC: "true" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - PYTHONUNBUFFERED: "1" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - - # KV cache and attention - attention-backend: "trtllm_mla" - kv-cache-dtype: "fp8_e4m3" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - stream-interval: 50 - max-running-requests: 30000 - context-length: 9300 - watchdog-timeout: 1000000 - disable-shared-experts-fusion: true - eplb-algorithm: "deepseek" - - # Prefill-specific mode - disaggregation-mode: "prefill" - - # Memory and token limits - mem-fraction-static: 0.80 - max-total-tokens: 524288 - chunked-prefill-size: 131072 - - # Request handling - load-balance-method: "round_robin" - - # Performance optimizations - disable-cuda-graph: true - - # DeepEP configuration - moe-a2a-backend: "deepep" - deepep-mode: "normal" - ep-dispatch-algorithm: "dynamic" - moe-dense-tp-size: 1 - enable-dp-lm-head: true - ep-num-redundant-experts: 32 - deepep-config: "/configs/deepep_config.json" - disaggregation-bootstrap-port: 30001 - disaggregation-transfer-backend: nixl - - decode: - nodes: 6 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - MC_TE_METRIC: "true" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - PYTHONUNBUFFERED: "1" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 24 - dp-size: 24 - ep-size: 24 - enable-dp-attention: true - - # KV cache and attention - attention-backend: "trtllm_mla" - kv-cache-dtype: "fp8_e4m3" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - stream-interval: 50 - decode-log-interval: 1000 - max-running-requests: 8192 - context-length: 9300 - watchdog-timeout: 1000000 - disable-shared-experts-fusion: true - eplb-algorithm: "deepseek" - - # Decode-specific mode - disaggregation-mode: "decode" - - # Memory and token limits - mem-fraction-static: 0.82 - chunked-prefill-size: 36864 - - # DeepEP configuration - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - moe-dense-tp-size: 1 - enable-dp-lm-head: true - prefill-round-robin-balance: true - ep-num-redundant-experts: 32 - deepep-config: "/configs/deepep_config.json" - # CUDA graphs - cuda-graph-bs: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, - 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 264, 272, 280, 288, 296, 304, 312, 320, 328, 336, - 344, 352, 360, 368, 376, 384, 416, 448, 480, 512] - cuda-graph-max-bs: 512 - disaggregation-bootstrap-port: 30001 - disaggregation-transfer-backend: nixl - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "2048x4096x6144" - req_rate: "300" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp8/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp8/8k1k/variants.yaml new file mode 100644 index 0000000000..e5600451ac --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb200-fp8/8k1k/variants.yaml @@ -0,0 +1,183 @@ +# srt-slurm recipes for dsr1/sglang/gb200-fp8/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/sglang/gb200-fp8/8k1k/variants.yaml:override_. + +schema: 2 + +base: + dynamo: + request_plane: nats + source: + pypi: 0.8.1 + # Use NATS for a recipe prior to Dynamo commit 39d2a68. + frontend: + type: dynamo + enable_multiple_frontends: true + num_additional_frontends: 9 + nginx_container: nginx + model: + path: dsr1-fp8 + container: dynamo-sglang + precision: fp8 + resources: + gpu_type: gb200 + gpus_per_node: 4 + engine: sglang + roles: + prefill: + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + MC_TE_METRIC: 'true' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + PYTHONUNBUFFERED: '1' + # Decode-specific environment variables + args: + # Model configuration + served-model-name: deepseek-ai/DeepSeek-R1 + skip-tokenizer-init: true + trust-remote-code: true + # Parallelism + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + # KV cache and attention + attention-backend: trtllm_mla + kv-cache-dtype: fp8_e4m3 + # Radix cache disabled + disable-radix-cache: true + # Other flags + stream-interval: 50 + max-running-requests: 30000 + context-length: 9300 + watchdog-timeout: 1000000 + disable-shared-experts-fusion: true + eplb-algorithm: deepseek + # Prefill-specific mode + disaggregation-mode: prefill + # Memory and token limits + mem-fraction-static: 0.8 + max-total-tokens: 524288 + chunked-prefill-size: 131072 + # Request handling + load-balance-method: round_robin + # Performance optimizations + disable-cuda-graph: true + # DeepEP configuration + moe-a2a-backend: deepep + deepep-mode: normal + ep-dispatch-algorithm: dynamic + moe-dense-tp-size: 1 + enable-dp-lm-head: true + ep-num-redundant-experts: 32 + deepep-config: /configs/deepep_config.json + disaggregation-bootstrap-port: 30001 + disaggregation-transfer-backend: nixl + decode: + workers: 1 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + MC_TE_METRIC: 'true' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + PYTHONUNBUFFERED: '1' + # Parallelism + args: + # Model configuration + served-model-name: deepseek-ai/DeepSeek-R1 + skip-tokenizer-init: true + trust-remote-code: true + enable-dp-attention: true + # KV cache and attention + attention-backend: trtllm_mla + kv-cache-dtype: fp8_e4m3 + # Radix cache disabled + disable-radix-cache: true + # Other flags + stream-interval: 50 + decode-log-interval: 1000 + max-running-requests: 8192 + context-length: 9300 + watchdog-timeout: 1000000 + disable-shared-experts-fusion: true + eplb-algorithm: deepseek + # Decode-specific mode + disaggregation-mode: decode + # Memory and token limits + mem-fraction-static: 0.82 + chunked-prefill-size: 36864 + # DeepEP configuration + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + moe-dense-tp-size: 1 + enable-dp-lm-head: true + prefill-round-robin-balance: true + ep-num-redundant-experts: 32 + deepep-config: /configs/deepep_config.json + disaggregation-bootstrap-port: 30001 + disaggregation-transfer-backend: nixl + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: '300' + +override_disagg_5p1d_p_tp8_ep8_dp8_d_tp32_ep32_dp32_b8192_c512x1024x2048x6144_stp: + name: gb200-8k1k-fp8-mid-tpt + roles: + prefill: + nodes: 10 + workers: 5 + decode: + nodes: 8 + env: + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' + args: + tp-size: 32 + dp-size: 32 + ep-size: 32 + # CUDA graphs + cuda-graph-max-bs: 256 + benchmark: + concurrencies: 512x1024x2048x6144 + +override_disagg_6p1d_p_tp8_ep8_dp8_d_tp24_ep24_dp24_b8192_c2048x4096x6144_stp: + name: gb200-8k1k-fp8-max-tpt + roles: + prefill: + nodes: 12 + workers: 6 + decode: + nodes: 6 + env: + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tp-size: 24 + dp-size: 24 + ep-size: 24 + cuda-graph-max-bs: 512 + # CUDA graphs + cuda-graph-bs: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 264, 272, 280, 288, 296, 304, 312, 320, 328, 336, 344, 352, 360, 368, 376, 384, 416, 448, 480, 512] + benchmark: + concurrencies: 2048x4096x6144 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-10p1d-p-tp4-d-tp32-ep32-dp32-b16384-c2048-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-10p1d-p-tp4-d-tp32-ep32-dp32-b16384-c2048-stp.yaml deleted file mode 100644 index f7b29609f5..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-10p1d-p-tp4-d-tp32-ep32-dp32-b16384-c2048-stp.yaml +++ /dev/null @@ -1,182 +0,0 @@ -schema: 2 -name: "gb300-fp4-8k1k-max-tpt" - -dynamo: - request_plane: "nats" - source: - pypi: 0.8.1 - # Use NATS for a recipe prior to Dynamo commit 39d2a68. -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 9 - nginx_container: nginx-sqsh - -model: - path: "dsr1" - container: "dynamo-sglang" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: sglang -roles: - prefill: - nodes: 10 - workers: 10 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: "1" - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - MC_TE_METRIC: "true" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - trust-remote-code: true - - # KV cache and attention - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - - # Quantization - quantization: "modelopt_fp4" - moe-runner-backend: "flashinfer_trtllm" - - # Radix cache disabled - disable-radix-cache: true - disable-chunked-prefix-cache: true - - # Other flags - stream-interval: 50 - decode-log-interval: 1000 - watchdog-timeout: 1000000 - context-length: 9600 - disable-shared-experts-fusion: true - disaggregation-bootstrap-port: 30001 - - # Prefill-specific mode - disaggregation-mode: "prefill" - - # Memory and token limits - mem-fraction-static: 0.95 - max-total-tokens: 131072 - max-prefill-tokens: 524288 - chunked-prefill-size: 131072 - - # Request handling - max-running-requests: 30000 - load-balance-method: "round_robin" - - # Performance optimizations - disable-cuda-graph: true - enable-dp-attention: false - fp4-gemm-backend: "flashinfer_cutlass" - disaggregation-transfer-backend: nixl - - # Parallelism - tp-size: 4 - dp-size: 1 - ep-size: 1 - - decode: - nodes: 8 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: "1" - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - MC_TE_METRIC: "true" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_MOE_NVFP4_DISPATCH: "1" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - trust-remote-code: true - - # KV cache and attention - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - - # Quantization - quantization: "modelopt_fp4" - moe-runner-backend: "flashinfer_cutedsl" - - # Radix cache disabled - disable-radix-cache: true - disable-chunked-prefix-cache: true - - # Other flags - stream-interval: 50 - decode-log-interval: 1000 - watchdog-timeout: 1000000 - context-length: 9600 - disable-shared-experts-fusion: true - eplb-algorithm: "deepseek" - disaggregation-bootstrap-port: 30001 - - # Decode-specific mode - disaggregation-mode: "decode" - - # Memory and token limits - mem-fraction-static: 0.83 - max-total-tokens: 524288 - chunked-prefill-size: 24576 - - # Request handling - max-running-requests: 16384 - - # DeepEP configuration - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - ep-num-redundant-experts: 32 - - cuda-graph-max-bs: 512 - num-reserved-decode-tokens: 112 - - # Additional decode optimizations - moe-dense-tp-size: 1 - enable-dp-lm-head: true - prefill-round-robin-balance: true - enable-dp-attention: true - fp4-gemm-backend: "flashinfer_cutlass" - disaggregation-transfer-backend: nixl - - # Parallelism - tp-size: 32 - dp-size: 32 - ep-size: 32 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "2048" - req_rate: 700 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-1p4d-p-tp4-d-tp4-c4x8x32x64-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-1p4d-p-tp4-d-tp4-c4x8x32x64-stp.yaml deleted file mode 100644 index d150e71670..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-1p4d-p-tp4-d-tp4-c4x8x32x64-stp.yaml +++ /dev/null @@ -1,124 +0,0 @@ -schema: 2 -name: "gb300-8k1k-fp4-low-latency-8k1k" - -dynamo: - request_plane: "nats" - - source: - pypi: 0.8.1 - # Use NATS for a recipe prior to Dynamo commit 39d2a68. -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 3 - nginx_container: nginx-sqsh - -model: - path: "dsr1" - container: "dynamo-sglang" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_ENABLE_JIT_DEEPGEMM: "false" - - args: - disaggregation-mode: "prefill" - served-model-name: "deepseek-ai/DeepSeek-R1" - trust-remote-code: true - disable-radix-cache: true - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - quantization: "modelopt_fp4" - moe-runner-backend: "flashinfer_trtllm" - stream-interval: 50 - watchdog-timeout: 1000000 - context-length: 9600 - mem-fraction-static: 0.95 - max-total-tokens: 32768 - chunked-prefill-size: 24576 - cuda-graph-max-bs: 256 - max-running-requests: 512 - scheduler-recv-interval: 10 - enable-symm-mem: true - load-balance-method: "round_robin" - disaggregation-bootstrap-port: 30001 - data-parallel-size: 1 - tensor-parallel-size: 4 - expert-parallel-size: 1 - enable-dp-attention: false - fp4-gemm-backend: "flashinfer_trtllm" - disaggregation-transfer-backend: nixl - - - decode: - nodes: 4 - workers: 4 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_ENABLE_JIT_DEEPGEMM: "false" - - args: - disaggregation-mode: "decode" - served-model-name: "deepseek-ai/DeepSeek-R1" - prefill-round-robin-balance: true - trust-remote-code: true - disable-radix-cache: true - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - quantization: "modelopt_fp4" - moe-runner-backend: "flashinfer_trtllm" - disaggregation-bootstrap-port: 30001 - stream-interval: 50 - watchdog-timeout: 1000000 - context-length: 9600 - mem-fraction-static: 0.95 - chunked-prefill-size: 8192 - cuda-graph-max-bs: 128 - scheduler-recv-interval: 10 - enable-symm-mem: true - tensor-parallel-size: 4 - expert-parallel-size: 1 - enable-dp-attention: false - fp4-gemm-backend: "flashinfer_trtllm" - disaggregation-transfer-backend: nixl - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "4x8x32x64" - req_rate: 300 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-6p1d-p-tp4-d-tp48-ep48-dp48-b16384-c512x2048x4096-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-6p1d-p-tp4-d-tp48-ep48-dp48-b16384-c512x2048x4096-stp.yaml deleted file mode 100644 index e83bebe1cc..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-6p1d-p-tp4-d-tp48-ep48-dp48-b16384-c512x2048x4096-stp.yaml +++ /dev/null @@ -1,183 +0,0 @@ -schema: 2 -name: "gb300-fp4-8k1k-mid-curve" - -dynamo: - request_plane: "nats" - - source: - pypi: 0.8.1 - # Use NATS for a recipe prior to Dynamo commit 39d2a68. -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 9 - nginx_container: nginx-sqsh - -model: - path: "dsr1" - container: "dynamo-sglang" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: sglang -roles: - prefill: - nodes: 6 - workers: 6 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: "1" - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - MC_TE_METRIC: "true" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - trust-remote-code: true - - # KV cache and attention - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - - # Quantization - quantization: "modelopt_fp4" - moe-runner-backend: "flashinfer_trtllm" - - # Radix cache disabled - disable-radix-cache: true - disable-chunked-prefix-cache: true - - # Other flags - stream-interval: 50 - decode-log-interval: 1000 - watchdog-timeout: 1000000 - context-length: 9600 - disable-shared-experts-fusion: true - disaggregation-bootstrap-port: 30001 - - # Prefill-specific mode - disaggregation-mode: "prefill" - - # Memory and token limits - mem-fraction-static: 0.95 - max-total-tokens: 131072 - max-prefill-tokens: 524288 - chunked-prefill-size: 131072 - - # Request handling - max-running-requests: 30000 - load-balance-method: "round_robin" - - # Performance optimizations - disable-cuda-graph: true - enable-dp-attention: false - fp4-gemm-backend: "flashinfer_cutlass" - disaggregation-transfer-backend: nixl - - # Parallelism - tp-size: 4 - dp-size: 1 - ep-size: 1 - - decode: - nodes: 12 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: "1" - SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - MC_TE_METRIC: "true" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_MOE_NVFP4_DISPATCH: "1" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - trust-remote-code: true - - # KV cache and attention - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - - # Quantization - quantization: "modelopt_fp4" - moe-runner-backend: "flashinfer_cutedsl" - - # Radix cache disabled - disable-radix-cache: true - disable-chunked-prefix-cache: true - - # Other flags - stream-interval: 50 - decode-log-interval: 1000 - watchdog-timeout: 1000000 - context-length: 9600 - disable-shared-experts-fusion: true - eplb-algorithm: "deepseek" - disaggregation-bootstrap-port: 30001 - - # Decode-specific mode - disaggregation-mode: "decode" - - # Memory and token limits - mem-fraction-static: 0.83 - max-total-tokens: 524288 - chunked-prefill-size: 24576 - - # Request handling - max-running-requests: 16384 - - # DeepEP configuration - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - ep-num-redundant-experts: 32 - - cuda-graph-max-bs: 512 - num-reserved-decode-tokens: 112 - - # Additional decode optimizations - moe-dense-tp-size: 1 - enable-dp-lm-head: true - prefill-round-robin-balance: true - enable-dp-attention: true - fp4-gemm-backend: "flashinfer_cutlass" - disaggregation-transfer-backend: nixl - - # Parallelism - tp-size: 48 - dp-size: 48 - ep-size: 48 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "512x2048x4096" - req_rate: 700 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp4/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp4/8k1k/variants.yaml new file mode 100644 index 0000000000..50de2430d0 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp4/8k1k/variants.yaml @@ -0,0 +1,297 @@ +# srt-slurm recipes for dsr1/sglang/gb300-fp4/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/sglang/gb300-fp4/8k1k/variants.yaml:override_. + +schema: 2 + +base: + dynamo: + request_plane: nats + source: + pypi: 0.8.1 + # Use NATS for a recipe prior to Dynamo commit 39d2a68. + frontend: + type: dynamo + enable_multiple_frontends: true + nginx_container: nginx-sqsh + model: + path: dsr1 + container: dynamo-sglang + precision: fp4 + resources: + gpu_type: gb300 + gpus_per_node: 4 + engine: sglang + roles: + prefill: + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + args: + served-model-name: deepseek-ai/DeepSeek-R1 + trust-remote-code: true + kv-cache-dtype: fp8_e4m3 + attention-backend: trtllm_mla + quantization: modelopt_fp4 + moe-runner-backend: flashinfer_trtllm + disable-radix-cache: true + stream-interval: 50 + watchdog-timeout: 1000000 + context-length: 9600 + disaggregation-bootstrap-port: 30001 + disaggregation-mode: prefill + mem-fraction-static: 0.95 + load-balance-method: round_robin + enable-dp-attention: false + disaggregation-transfer-backend: nixl + decode: + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + args: + served-model-name: deepseek-ai/DeepSeek-R1 + trust-remote-code: true + kv-cache-dtype: fp8_e4m3 + attention-backend: trtllm_mla + quantization: modelopt_fp4 + disable-radix-cache: true + stream-interval: 50 + watchdog-timeout: 1000000 + context-length: 9600 + disaggregation-bootstrap-port: 30001 + disaggregation-mode: decode + prefill-round-robin-balance: true + disaggregation-transfer-backend: nixl + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + +override_disagg_10p1d_p_tp4_d_tp32_ep32_dp32_b16384_c2048_stp: + name: gb300-fp4-8k1k-max-tpt + frontend: + num_additional_frontends: 9 + roles: + prefill: + nodes: 10 + workers: 10 + env: + SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: '1' + SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' + MC_TE_METRIC: 'true' + # Decode-specific environment variables + # Model configuration + # KV cache and attention + # Quantization + # Radix cache disabled + # Other flags + # Prefill-specific mode + # Memory and token limits + args: + disable-chunked-prefix-cache: true + decode-log-interval: 1000 + disable-shared-experts-fusion: true + max-total-tokens: 131072 + max-prefill-tokens: 524288 + chunked-prefill-size: 131072 + # Request handling + max-running-requests: 30000 + # Performance optimizations + disable-cuda-graph: true + fp4-gemm-backend: flashinfer_cutlass + # Parallelism + tp-size: 4 + dp-size: 1 + ep-size: 1 + decode: + nodes: 8 + workers: 1 + env: + SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: '1' + SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' + MC_TE_METRIC: 'true' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + SGLANG_MOE_NVFP4_DISPATCH: '1' + # Model configuration + # KV cache and attention + # Quantization + # Radix cache disabled + # Other flags + # Decode-specific mode + args: + moe-runner-backend: flashinfer_cutedsl + disable-chunked-prefix-cache: true + decode-log-interval: 1000 + disable-shared-experts-fusion: true + eplb-algorithm: deepseek + # Memory and token limits + mem-fraction-static: 0.83 + max-total-tokens: 524288 + chunked-prefill-size: 24576 + # Request handling + max-running-requests: 16384 + # DeepEP configuration + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + ep-num-redundant-experts: 32 + cuda-graph-max-bs: 512 + num-reserved-decode-tokens: 112 + # Additional decode optimizations + moe-dense-tp-size: 1 + enable-dp-lm-head: true + enable-dp-attention: true + fp4-gemm-backend: flashinfer_cutlass + # Parallelism + tp-size: 32 + dp-size: 32 + ep-size: 32 + benchmark: + concurrencies: '2048' + req_rate: 700 + +override_disagg_1p4d_p_tp4_d_tp4_c4x8x32x64_stp: + name: gb300-8k1k-fp4-low-latency-8k1k + frontend: + num_additional_frontends: 3 + roles: + prefill: + nodes: 1 + workers: 1 + env: + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_ENABLE_JIT_DEEPGEMM: 'false' + args: + max-total-tokens: 32768 + chunked-prefill-size: 24576 + max-running-requests: 512 + fp4-gemm-backend: flashinfer_trtllm + cuda-graph-max-bs: 256 + scheduler-recv-interval: 10 + enable-symm-mem: true + data-parallel-size: 1 + tensor-parallel-size: 4 + expert-parallel-size: 1 + decode: + nodes: 4 + workers: 4 + env: + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_ENABLE_JIT_DEEPGEMM: 'false' + args: + moe-runner-backend: flashinfer_trtllm + mem-fraction-static: 0.95 + chunked-prefill-size: 8192 + cuda-graph-max-bs: 128 + enable-dp-attention: false + fp4-gemm-backend: flashinfer_trtllm + scheduler-recv-interval: 10 + enable-symm-mem: true + tensor-parallel-size: 4 + expert-parallel-size: 1 + benchmark: + concurrencies: 4x8x32x64 + req_rate: 300 + +override_disagg_6p1d_p_tp4_d_tp48_ep48_dp48_b16384_c512x2048x4096_stp: + name: gb300-fp4-8k1k-mid-curve + frontend: + num_additional_frontends: 9 + roles: + prefill: + nodes: 6 + workers: 6 + env: + SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: '1' + SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' + MC_TE_METRIC: 'true' + # Decode-specific environment variables + # Model configuration + # KV cache and attention + # Quantization + # Radix cache disabled + # Other flags + # Prefill-specific mode + # Memory and token limits + args: + disable-chunked-prefix-cache: true + decode-log-interval: 1000 + disable-shared-experts-fusion: true + max-total-tokens: 131072 + max-prefill-tokens: 524288 + chunked-prefill-size: 131072 + # Request handling + max-running-requests: 30000 + # Performance optimizations + disable-cuda-graph: true + fp4-gemm-backend: flashinfer_cutlass + # Parallelism + tp-size: 4 + dp-size: 1 + ep-size: 1 + decode: + nodes: 12 + workers: 1 + env: + SGLANG_NVFP4_CKPT_FP8_GEMM_IN_ATTN: '1' + SGLANG_PER_TOKEN_GROUP_QUANT_8BIT_V2: '1' + MC_TE_METRIC: 'true' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + SGLANG_MOE_NVFP4_DISPATCH: '1' + # Model configuration + # KV cache and attention + # Quantization + # Radix cache disabled + # Other flags + # Decode-specific mode + args: + moe-runner-backend: flashinfer_cutedsl + disable-chunked-prefix-cache: true + decode-log-interval: 1000 + disable-shared-experts-fusion: true + eplb-algorithm: deepseek + # Memory and token limits + mem-fraction-static: 0.83 + max-total-tokens: 524288 + chunked-prefill-size: 24576 + # Request handling + max-running-requests: 16384 + # DeepEP configuration + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + ep-num-redundant-experts: 32 + cuda-graph-max-bs: 512 + num-reserved-decode-tokens: 112 + # Additional decode optimizations + moe-dense-tp-size: 1 + enable-dp-lm-head: true + enable-dp-attention: true + fp4-gemm-backend: flashinfer_cutlass + # Parallelism + tp-size: 48 + dp-size: 48 + ep-size: 48 + benchmark: + concurrencies: 512x2048x4096 + req_rate: 700 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c4x8-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c4x8-stp.yaml deleted file mode 100644 index e339b0137a..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c4x8-stp.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: "gb300-8k1k-fp8-low-latency" - -model: - path: "dsr1-fp8" - container: "dynamo-sglang" - precision: "fp8" - -frontend: - nginx_container: nginx - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -slurm: - time_limit: "02:00:00" - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_DG_CACHE_DIR: "/configs/dg-10212025" - SGLANG_ENABLE_JIT_DEEPGEMM: "false" - # SGLANG_ENABLE_FLASHINFER_GEMM: "1" # deprecated in 0.5.7, --fp8-gemm-backend=flashinfer_trtllm - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - MC_TE_METRIC: "true" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - - args: - served-model-name: "deepseek-ai/DeepSeek-R1" - trust-remote-code: true - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - quantization: "fp8" - moe-runner-backend: "flashinfer_trtllm" - fp8-gemm-backend: "flashinfer_trtllm" - disable-radix-cache: true - stream-interval: 10 - watchdog-timeout: 1000000 - context-length: 9300 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - mem-fraction-static: 0.95 - max-total-tokens: 32768 - chunked-prefill-size: 32768 - max-prefill-tokens: 32768 - cuda-graph-max-bs: 128 - max-running-requests: 128 - load-balance-method: "round_robin" - scheduler-recv-interval: 10 - enable-flashinfer-allreduce-fusion: false # to save mem - enable-symm-mem: false # to save mem - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - - decode: - nodes: 1 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - PYTHONUNBUFFERED: "1" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_DG_CACHE_DIR: "/configs/dg-10212025" - SGLANG_ENABLE_JIT_DEEPGEMM: "false" - # SGLANG_ENABLE_FLASHINFER_GEMM: "1" # deprecated in 0.5.7, --fp8-gemm-backend=flashinfer_trtllm - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - MC_TE_METRIC: "true" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - - args: - served-model-name: "deepseek-ai/DeepSeek-R1" - trust-remote-code: true - kv-cache-dtype: "fp8_e4m3" - attention-backend: "trtllm_mla" - quantization: "fp8" - moe-runner-backend: "flashinfer_trtllm" - fp8-gemm-backend: "flashinfer_trtllm" - disable-radix-cache: true - stream-interval: 10 - watchdog-timeout: 1000000 - context-length: 9300 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - mem-fraction-static: 0.85 - chunked-prefill-size: -1 # save mem - cuda-graph-max-bs: 128 - max-running-requests: 128 - scheduler-recv-interval: 1 # save mem - enable-flashinfer-allreduce-fusion: false # to save mem - enable-symm-mem: false # to save mem - prefill-round-robin-balance: true - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [4, 8] - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-5p1d-p-tp8-ep8-dp8-d-tp32-ep32-dp32-b45000-c128x256x512x1024-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-5p1d-p-tp8-ep8-dp8-d-tp32-ep32-dp32-b45000-c128x256x512x1024-stp.yaml deleted file mode 100644 index 88f207b6e7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-5p1d-p-tp8-ep8-dp8-d-tp32-ep32-dp32-b45000-c128x256x512x1024-stp.yaml +++ /dev/null @@ -1,180 +0,0 @@ -# GB300 FP8 Mid Throughput Configuration - -schema: 2 -name: "gb300-8k1k-fp8-mid" - -model: - path: "dsr1-fp8" - container: "dynamo-sglang" - precision: "fp8" - -frontend: - nginx_container: nginx - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -engine: sglang -roles: - prefill: - nodes: 10 - workers: 5 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_DG_CACHE_DIR: "/configs/dg-10212025" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - MC_TE_METRIC: "true" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - PYTHONUNBUFFERED: "1" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - - # KV cache and attention - attention-backend: "trtllm_mla" - kv-cache-dtype: "fp8_e4m3" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - stream-interval: 50 - max-running-requests: 30000 - context-length: 9300 - watchdog-timeout: 1000000 - disable-shared-experts-fusion: true - eplb-algorithm: "deepseek" - disaggregation-bootstrap-port: 30001 - disaggregation-transfer-backend: nixl - - # Prefill-specific mode - disaggregation-mode: "prefill" - - # Memory and token limits - mem-fraction-static: 0.75 - max-total-tokens: 524288 - chunked-prefill-size: 131072 - - # Request handling - load-balance-method: "round_robin" - - # Performance optimizations - disable-cuda-graph: true - - # DeepEP configuration - moe-a2a-backend: "deepep" - deepep-mode: "normal" - ep-dispatch-algorithm: "dynamic" - moe-dense-tp-size: 1 - enable-dp-lm-head: true - ep-num-redundant-experts: 32 - deepep-config: "/configs/deepep_config.json" - - decode: - nodes: 8 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_DG_CACHE_DIR: "/configs/dg-10212025" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "768" - MC_TE_METRIC: "true" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - PYTHONUNBUFFERED: "1" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - skip-tokenizer-init: true - trust-remote-code: true - disaggregation-transfer-backend: nixl - - # Parallelism - tp-size: 32 - dp-size: 32 - ep-size: 32 - enable-dp-attention: true - - # KV cache and attention - attention-backend: "trtllm_mla" - kv-cache-dtype: "fp8_e4m3" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - stream-interval: 50 - decode-log-interval: 1000 - max-running-requests: 45000 - context-length: 9300 - - watchdog-timeout: 1000000 - disable-shared-experts-fusion: true - eplb-algorithm: "deepseek" - disaggregation-bootstrap-port: 30001 - - # Decode-specific mode - disaggregation-mode: "decode" - - # Memory and token limits - mem-fraction-static: 0.82 - chunked-prefill-size: 36864 - - # DeepEP configuration - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - moe-dense-tp-size: 1 - enable-dp-lm-head: true - prefill-round-robin-balance: true - ep-num-redundant-experts: 32 - deepep-config: "/configs/deepep_config.json" - - # CUDA graphs - cuda-graph-bs: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, - 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 264, 272, 280, 288, 296, 304, 312, 320, 328, 336, - 344, 352, 360, 368, 376, 384, 416, 448, 480, 512, 544, 576, 608, 640, 672, 704, 736, 768] - cuda-graph-max-bs: 768 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [128, 256, 512, 1024] - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp8-ep8-dp8-d-tp24-ep24-dp24-b45000-c2048x4096-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp8-ep8-dp8-d-tp24-ep24-dp24-b45000-c2048x4096-stp.yaml deleted file mode 100644 index c8bf36e34a..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp8-ep8-dp8-d-tp24-ep24-dp24-b45000-c2048x4096-stp.yaml +++ /dev/null @@ -1,180 +0,0 @@ -# GB300 FP8 Max Throughput Configuration - -schema: 2 -name: "gb300-8k1k-fp8-max" - -model: - path: "dsr1-fp8" - container: "dynamo-sglang" - precision: "fp8" - -frontend: - nginx_container: nginx - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -engine: sglang -roles: - prefill: - nodes: 12 - workers: 6 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_DG_CACHE_DIR: "/configs/dg-10212025" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - MC_TE_METRIC: "true" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - PYTHONUNBUFFERED: "1" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - - # KV cache and attention - attention-backend: "trtllm_mla" - kv-cache-dtype: "fp8_e4m3" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - stream-interval: 50 - max-running-requests: 30000 - context-length: 9300 - watchdog-timeout: 1000000 - disable-shared-experts-fusion: true - eplb-algorithm: "deepseek" - disaggregation-bootstrap-port: 30001 - disaggregation-transfer-backend: nixl - - # Prefill-specific mode - disaggregation-mode: "prefill" - - # Memory and token limits - mem-fraction-static: 0.75 - max-total-tokens: 524288 - chunked-prefill-size: 131072 - - # Request handling - load-balance-method: "round_robin" - - # Performance optimizations - disable-cuda-graph: true - - # DeepEP configuration - moe-a2a-backend: "deepep" - deepep-mode: "normal" - ep-dispatch-algorithm: "dynamic" - moe-dense-tp-size: 1 - enable-dp-lm-head: true - ep-num-redundant-experts: 32 - deepep-config: "/configs/deepep_config.json" - - decode: - nodes: 6 - workers: 1 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_DG_CACHE_DIR: "/configs/dg-10212025" - DYN_SKIP_SGLANG_LOG_FORMATTING: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "768" - MC_TE_METRIC: "true" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - MC_FORCE_MNNVL: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - PYTHONUNBUFFERED: "1" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - skip-tokenizer-init: true - trust-remote-code: true - disaggregation-transfer-backend: nixl - - # Parallelism - tp-size: 24 - dp-size: 24 - ep-size: 24 - enable-dp-attention: true - - # KV cache and attention - attention-backend: "trtllm_mla" - kv-cache-dtype: "fp8_e4m3" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - stream-interval: 50 - decode-log-interval: 1000 - max-running-requests: 45000 - context-length: 9300 - - watchdog-timeout: 1000000 - disable-shared-experts-fusion: true - eplb-algorithm: "deepseek" - disaggregation-bootstrap-port: 30001 - - # Decode-specific mode - disaggregation-mode: "decode" - - # Memory and token limits - mem-fraction-static: 0.82 - chunked-prefill-size: 36864 - - # DeepEP configuration - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - moe-dense-tp-size: 1 - enable-dp-lm-head: true - prefill-round-robin-balance: true - ep-num-redundant-experts: 32 - deepep-config: "/configs/deepep_config.json" - - # CUDA graphs - cuda-graph-bs: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, - 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 264, 272, 280, 288, 296, 304, 312, 320, 328, 336, - 344, 352, 360, 368, 376, 384, 416, 448, 480, 512, 544, 576, 608, 640, 672, 704, 736, 768] - cuda-graph-max-bs: 768 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [2048, 4096] - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp8/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp8/8k1k/variants.yaml new file mode 100644 index 0000000000..ab179c33a5 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/gb300-fp8/8k1k/variants.yaml @@ -0,0 +1,301 @@ +# srt-slurm recipes for dsr1/sglang/gb300-fp8/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/sglang/gb300-fp8/8k1k/variants.yaml:override_. + +schema: 2 + +base: + model: + path: dsr1-fp8 + container: dynamo-sglang + precision: fp8 + frontend: + nginx_container: nginx + resources: + gpu_type: gb300 + gpus_per_node: 4 + dynamo: + # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. + # Use NATS for a recipe prior to Dynamo commit 39d2a68. + request_plane: nats + source: + pypi: 0.8.0 + engine: sglang + roles: + prefill: + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DG_CACHE_DIR: /configs/dg-10212025 + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + MC_TE_METRIC: 'true' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + args: + served-model-name: deepseek-ai/DeepSeek-R1 + trust-remote-code: true + kv-cache-dtype: fp8_e4m3 + attention-backend: trtllm_mla + disable-radix-cache: true + watchdog-timeout: 1000000 + context-length: 9300 + disaggregation-mode: prefill + disaggregation-transfer-backend: nixl + load-balance-method: round_robin + decode: + workers: 1 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DG_CACHE_DIR: /configs/dg-10212025 + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + MC_TE_METRIC: 'true' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + args: + served-model-name: deepseek-ai/DeepSeek-R1 + trust-remote-code: true + kv-cache-dtype: fp8_e4m3 + attention-backend: trtllm_mla + disable-radix-cache: true + watchdog-timeout: 1000000 + context-length: 9300 + disaggregation-mode: decode + disaggregation-transfer-backend: nixl + prefill-round-robin-balance: true + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + +override_disagg_1p1d_p_tp4_d_tp4_b128_c4x8_stp: + name: gb300-8k1k-fp8-low-latency + slurm: + time_limit: 02:00:00 + roles: + prefill: + nodes: 1 + workers: 1 + # SGLANG_ENABLE_FLASHINFER_GEMM: "1" # deprecated in 0.5.7, --fp8-gemm-backend=flashinfer_trtllm + env: + SGLANG_ENABLE_JIT_DEEPGEMM: 'false' + args: + quantization: fp8 + moe-runner-backend: flashinfer_trtllm + fp8-gemm-backend: flashinfer_trtllm + stream-interval: 10 + mem-fraction-static: 0.95 + max-total-tokens: 32768 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + cuda-graph-max-bs: 128 + max-running-requests: 128 + scheduler-recv-interval: 10 + # to save mem + enable-flashinfer-allreduce-fusion: false + # to save mem + enable-symm-mem: false + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + decode: + nodes: 1 + # SGLANG_ENABLE_FLASHINFER_GEMM: "1" # deprecated in 0.5.7, --fp8-gemm-backend=flashinfer_trtllm + env: + SGLANG_ENABLE_JIT_DEEPGEMM: 'false' + args: + quantization: fp8 + moe-runner-backend: flashinfer_trtllm + fp8-gemm-backend: flashinfer_trtllm + stream-interval: 10 + mem-fraction-static: 0.85 + # save mem + chunked-prefill-size: -1 + cuda-graph-max-bs: 128 + max-running-requests: 128 + # save mem + scheduler-recv-interval: 1 + # to save mem + enable-flashinfer-allreduce-fusion: false + # to save mem + enable-symm-mem: false + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + benchmark: + concurrencies: [4, 8] + +# GB300 FP8 Mid Throughput Configuration +override_disagg_5p1d_p_tp8_ep8_dp8_d_tp32_ep32_dp32_b45000_c128x256x512x1024_stp: + name: gb300-8k1k-fp8-mid + roles: + prefill: + nodes: 10 + workers: 5 + # Decode-specific environment variables + # Model configuration + # KV cache and attention + # Radix cache disabled + # Prefill-specific mode + # Request handling + args: + # Other flags + stream-interval: 50 + # Memory and token limits + mem-fraction-static: 0.75 + max-total-tokens: 524288 + chunked-prefill-size: 131072 + max-running-requests: 30000 + skip-tokenizer-init: true + # Parallelism + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + disable-shared-experts-fusion: true + eplb-algorithm: deepseek + disaggregation-bootstrap-port: 30001 + # Performance optimizations + disable-cuda-graph: true + # DeepEP configuration + moe-a2a-backend: deepep + deepep-mode: normal + ep-dispatch-algorithm: dynamic + moe-dense-tp-size: 1 + enable-dp-lm-head: true + ep-num-redundant-experts: 32 + deepep-config: /configs/deepep_config.json + decode: + nodes: 8 + env: + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '768' + # Model configuration + # KV cache and attention + # Radix cache disabled + # Decode-specific mode + args: + # Other flags + stream-interval: 50 + # Memory and token limits + mem-fraction-static: 0.82 + chunked-prefill-size: 36864 + cuda-graph-max-bs: 768 + max-running-requests: 45000 + skip-tokenizer-init: true + # Parallelism + tp-size: 32 + dp-size: 32 + ep-size: 32 + enable-dp-attention: true + decode-log-interval: 1000 + disable-shared-experts-fusion: true + eplb-algorithm: deepseek + disaggregation-bootstrap-port: 30001 + # DeepEP configuration + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + moe-dense-tp-size: 1 + enable-dp-lm-head: true + ep-num-redundant-experts: 32 + deepep-config: /configs/deepep_config.json + # CUDA graphs + cuda-graph-bs: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 264, 272, 280, 288, 296, 304, 312, 320, 328, 336, 344, 352, 360, 368, 376, 384, 416, 448, 480, 512, 544, 576, 608, 640, 672, 704, 736, 768] + benchmark: + concurrencies: [128, 256, 512, 1024] + +# GB300 FP8 Max Throughput Configuration +override_disagg_6p1d_p_tp8_ep8_dp8_d_tp24_ep24_dp24_b45000_c2048x4096_stp: + name: gb300-8k1k-fp8-max + roles: + prefill: + nodes: 12 + workers: 6 + # Decode-specific environment variables + # Model configuration + # KV cache and attention + # Radix cache disabled + # Prefill-specific mode + # Request handling + args: + # Other flags + stream-interval: 50 + # Memory and token limits + mem-fraction-static: 0.75 + max-total-tokens: 524288 + chunked-prefill-size: 131072 + max-running-requests: 30000 + skip-tokenizer-init: true + # Parallelism + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + disable-shared-experts-fusion: true + eplb-algorithm: deepseek + disaggregation-bootstrap-port: 30001 + # Performance optimizations + disable-cuda-graph: true + # DeepEP configuration + moe-a2a-backend: deepep + deepep-mode: normal + ep-dispatch-algorithm: dynamic + moe-dense-tp-size: 1 + enable-dp-lm-head: true + ep-num-redundant-experts: 32 + deepep-config: /configs/deepep_config.json + decode: + nodes: 6 + env: + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '768' + # Model configuration + # KV cache and attention + # Radix cache disabled + # Decode-specific mode + args: + # Other flags + stream-interval: 50 + # Memory and token limits + mem-fraction-static: 0.82 + chunked-prefill-size: 36864 + cuda-graph-max-bs: 768 + max-running-requests: 45000 + skip-tokenizer-init: true + # Parallelism + tp-size: 24 + dp-size: 24 + ep-size: 24 + enable-dp-attention: true + decode-log-interval: 1000 + disable-shared-experts-fusion: true + eplb-algorithm: deepseek + disaggregation-bootstrap-port: 30001 + # DeepEP configuration + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + moe-dense-tp-size: 1 + enable-dp-lm-head: true + ep-num-redundant-experts: 32 + deepep-config: /configs/deepep_config.json + # CUDA graphs + cuda-graph-bs: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 264, 272, 280, 288, 296, 304, 312, 320, 328, 336, 344, 352, 360, 368, 376, 384, 416, 448, 480, 512, 544, 576, 608, 640, 672, 704, 736, 768] + benchmark: + concurrencies: [2048, 4096] diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-b128-c1x2x4x8x16x32x64x128-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-b128-c1x2x4x8x16x32x64x128-mtp.yaml deleted file mode 100644 index f08550fd32..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-b128-c1x2x4x8x16x32x64x128-mtp.yaml +++ /dev/null @@ -1,124 +0,0 @@ -schema: 2 -name: "h100-fp8-1p1d-max-tp-mtp" - -model: - path: "dsr1-fp8" - container: "lmsysorg/sglang:v0.5.8-cu130" - precision: "fp8" - -resources: - gpu_type: "h100" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -frontend: - nginx_container: nginx-sqsh - -engine: sglang -roles: - prefill: - nodes: 2 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_ENABLE_SPEC_V2: "1" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 16 - dp-size: 1 - ep-size: 1 - enable-dp-attention: false - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Prefill capacity - max-running-requests: 2 - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 2048 - chunked-prefill-size: 2048 - - # Request handling - load-balance-method: "round_robin" - - # MTP (Multi-Token Prediction) - speculative-algorithm: "EAGLE" - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - - decode: - nodes: 2 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_ENABLE_SPEC_V2: "1" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 16 - dp-size: 1 - ep-size: 1 - enable-dp-attention: false - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 1 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.9 - max-running-requests: 128 - cuda-graph-max-bs: 128 - - # MTP (Multi-Token Prediction) - speculative-algorithm: "EAGLE" - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1x2x4x8x16x32x64x128" - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-b128-c1x2x4x8x16x32x64x128-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-b128-c1x2x4x8x16x32x64x128-stp.yaml deleted file mode 100644 index e3507c45ca..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-b128-c1x2x4x8x16x32x64x128-stp.yaml +++ /dev/null @@ -1,110 +0,0 @@ -schema: 2 -name: "h100-fp8-1p1d-max-tp" - -model: - path: "dsr1-fp8" - container: "lmsysorg/sglang:v0.5.8-cu130" - precision: "fp8" - -resources: - gpu_type: "h100" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -frontend: - nginx_container: nginx-sqsh - -engine: sglang -roles: - prefill: - nodes: 2 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 16 - dp-size: 1 - ep-size: 1 - enable-dp-attention: false - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Prefill capacity - max-running-requests: 2 - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 2048 - chunked-prefill-size: 2048 - - # Request handling - load-balance-method: "round_robin" - - decode: - nodes: 2 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 16 - dp-size: 1 - ep-size: 1 - enable-dp-attention: false - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 1 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.9 - max-running-requests: 128 - cuda-graph-max-bs: 128 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1x2x4x8x16x32x64x128" - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-ep16-dp16-b64-c1x2x4x8x16x32x64-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-ep16-dp16-b64-c1x2x4x8x16x32x64-mtp.yaml deleted file mode 100644 index e6551f3905..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-ep16-dp16-b64-c1x2x4x8x16x32x64-mtp.yaml +++ /dev/null @@ -1,124 +0,0 @@ -schema: 2 -name: "h100-fp8-1p1d-max-dep-mtp" - -model: - path: "dsr1-fp8" - container: "lmsysorg/sglang:v0.5.8-cu130" - precision: "fp8" - -resources: - gpu_type: "h100" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -frontend: - nginx_container: nginx-sqsh - -engine: sglang -roles: - prefill: - nodes: 2 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_ENABLE_SPEC_V2: "1" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 16 - dp-size: 1 - ep-size: 1 - enable-dp-attention: false - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Prefill capacity - max-running-requests: 4 - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 2048 - chunked-prefill-size: 2048 - - # Request handling - load-balance-method: "round_robin" - - # MTP (Multi-Token Prediction) - speculative-algorithm: "EAGLE" - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - - decode: - nodes: 2 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_ENABLE_SPEC_V2: "1" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 16 - dp-size: 16 - ep-size: 16 - enable-dp-attention: true - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 1 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.85 - max-running-requests: 64 - cuda-graph-max-bs: 64 - - # MTP - speculative-algorithm: "EAGLE" - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1x2x4x8x16x32x64" - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-ep16-dp16-b64-c1x2x4x8x16x32x64-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-ep16-dp16-b64-c1x2x4x8x16x32x64-stp.yaml deleted file mode 100644 index 2ece076697..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-ep16-dp16-b64-c1x2x4x8x16x32x64-stp.yaml +++ /dev/null @@ -1,110 +0,0 @@ -schema: 2 -name: "h100-fp8-1p1d-max-dep" - -model: - path: "dsr1-fp8" - container: "lmsysorg/sglang:v0.5.8-cu130" - precision: "fp8" - -resources: - gpu_type: "h100" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -frontend: - nginx_container: nginx-sqsh - -engine: sglang -roles: - prefill: - nodes: 2 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 16 - dp-size: 1 - ep-size: 1 - enable-dp-attention: false - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Prefill capacity - max-running-requests: 4 - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.6 - max-prefill-tokens: 2048 - chunked-prefill-size: 2048 - - # Request handling - load-balance-method: "round_robin" - - decode: - nodes: 2 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 16 - dp-size: 16 - ep-size: 16 - enable-dp-attention: true - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 1 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.9 - max-running-requests: 64 - cuda-graph-max-bs: 64 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1x2x4x8x16x32x64" - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/variants.yaml new file mode 100644 index 0000000000..2ee82ec409 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h100-fp8/8k1k/variants.yaml @@ -0,0 +1,182 @@ +# srt-slurm recipes for dsr1/sglang/h100-fp8/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/sglang/h100-fp8/8k1k/variants.yaml:override_. + +schema: 2 + +base: + model: + path: dsr1-fp8 + container: lmsysorg/sglang:v0.5.8-cu130 + precision: fp8 + resources: + gpu_type: h100 + gpus_per_node: 8 + dynamo: + # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. + # Use NATS for a recipe prior to Dynamo commit 39d2a68. + request_plane: nats + source: + pypi: 0.8.0 + frontend: + nginx_container: nginx-sqsh + engine: sglang + roles: + prefill: + nodes: 2 + workers: 1 + env: + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + # Decode-specific environment variables + # Prefill capacity + args: + # Model configuration + served-model-name: deepseek-ai/DeepSeek-R1 + model-path: /model/ + skip-tokenizer-init: true + trust-remote-code: true + watchdog-timeout: 1000000 + # Parallelism + tp-size: 16 + dp-size: 1 + ep-size: 1 + enable-dp-attention: false + # KV cache and attention + attention-backend: flashinfer + # Radix cache disabled + disable-radix-cache: true + # Prefill-specific mode + disaggregation-bootstrap-port: 30001 + disaggregation-mode: prefill + disaggregation-transfer-backend: nixl + # Memory and token limits + mem-fraction-static: 0.6 + max-prefill-tokens: 2048 + chunked-prefill-size: 2048 + # Request handling + load-balance-method: round_robin + decode: + nodes: 2 + workers: 1 + env: + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + # Memory and token limits + args: + # Model configuration + served-model-name: deepseek-ai/DeepSeek-R1 + model-path: /model/ + skip-tokenizer-init: true + trust-remote-code: true + watchdog-timeout: 1000000 + # Parallelism + tp-size: 16 + # KV cache and attention + attention-backend: flashinfer + # Other flags + disable-radix-cache: true + stream-interval: 1 + # Disagg + disaggregation-bootstrap-port: 30001 + disaggregation-mode: decode + disaggregation-transfer-backend: nixl + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + +override_disagg_1p1d_p_tp16_d_tp16_b128_c1x2x4x8x16x32x64x128_mtp: + name: h100-fp8-1p1d-max-tp-mtp + roles: + prefill: + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + max-running-requests: 2 + # MTP (Multi-Token Prediction) + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + decode: + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + dp-size: 1 + ep-size: 1 + enable-dp-attention: false + mem-fraction-static: 0.9 + max-running-requests: 128 + cuda-graph-max-bs: 128 + # MTP (Multi-Token Prediction) + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + benchmark: + concurrencies: 1x2x4x8x16x32x64x128 + +override_disagg_1p1d_p_tp16_d_tp16_b128_c1x2x4x8x16x32x64x128_stp: + name: h100-fp8-1p1d-max-tp + roles: + prefill: + args: + max-running-requests: 2 + decode: + args: + dp-size: 1 + ep-size: 1 + enable-dp-attention: false + mem-fraction-static: 0.9 + max-running-requests: 128 + cuda-graph-max-bs: 128 + benchmark: + concurrencies: 1x2x4x8x16x32x64x128 + +override_disagg_1p1d_p_tp16_d_tp16_ep16_dp16_b64_c1x2x4x8x16x32x64_mtp: + name: h100-fp8-1p1d-max-dep-mtp + roles: + prefill: + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + max-running-requests: 4 + # MTP (Multi-Token Prediction) + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + decode: + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + dp-size: 16 + ep-size: 16 + enable-dp-attention: true + mem-fraction-static: 0.85 + max-running-requests: 64 + cuda-graph-max-bs: 64 + # MTP + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + benchmark: + concurrencies: 1x2x4x8x16x32x64 + +override_disagg_1p1d_p_tp16_d_tp16_ep16_dp16_b64_c1x2x4x8x16x32x64_stp: + name: h100-fp8-1p1d-max-dep + roles: + prefill: + args: + max-running-requests: 4 + decode: + args: + dp-size: 16 + ep-size: 16 + enable-dp-attention: true + mem-fraction-static: 0.9 + max-running-requests: 64 + cuda-graph-max-bs: 64 + benchmark: + concurrencies: 1x2x4x8x16x32x64 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs128-1p1d-dep-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs128-1p1d-dep-mtp.yaml deleted file mode 100644 index 50c761b46b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs128-1p1d-dep-mtp.yaml +++ /dev/null @@ -1,126 +0,0 @@ -schema: 2 -name: "bs128-1p1d-dep-h200-fp8-mtp" - -model: - path: "dsr1" - container: "lmsysorg/sglang:v0.5.8.post1-cu130" - precision: "fp8" - -frontend: - nginx_container: nginx - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - SGLANG_ENABLE_SPEC_V2: "1" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - # stream-interval: 50 - max-running-requests: 16 - - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - max-prefill-tokens: 163840 - chunked-prefill-size: 163840 - - # Request handling - load-balance-method: "round_robin" - - - decode: - nodes: 1 - workers: 1 - env: - SGLANG_ENABLE_SPEC_V2: "1" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 10 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.85 - max-running-requests: 192 - cuda-graph-max-bs: 192 - - # MTP settings - speculative-algorithm: "EAGLE" - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "32x64x128x256x512" - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs128-1p1d-dep-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs128-1p1d-dep-stp.yaml deleted file mode 100644 index bb0ddabd7f..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs128-1p1d-dep-stp.yaml +++ /dev/null @@ -1,117 +0,0 @@ -schema: 2 -name: "bs128-1p1d-dep-h200-fp8" - -model: - path: "dsr1" - container: "lmsysorg/sglang:v0.5.8.post1-cu130" - precision: "fp8" - -frontend: - nginx_container: nginx - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - # stream-interval: 50 - watchdog-timeout: 1000000 - max-running-requests: 16 - - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - max-prefill-tokens: 163840 - chunked-prefill-size: 163840 - - # Request handling - load-balance-method: "round_robin" - - - decode: - nodes: 1 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 10 - watchdog-timeout: 1000000 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.88 - max-running-requests: 256 - cuda-graph-max-bs: 256 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "64x128x256" - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs16-1p3d-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs16-1p3d-mtp.yaml deleted file mode 100644 index 1892e95fc1..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs16-1p3d-mtp.yaml +++ /dev/null @@ -1,124 +0,0 @@ -schema: 2 -name: "bs16-1p3d-h200-fp8-mtp" - -model: - path: "dsr1" - container: "lmsysorg/sglang:v0.5.8.post1-cu130" - precision: "fp8" - -frontend: - nginx_container: nginx - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - SGLANG_ENABLE_SPEC_V2: "1" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - # stream-interval: 50 - max-running-requests: 16 - - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.82 - max-prefill-tokens: 32768 - chunked-prefill-size: 32768 - - # Request handling - load-balance-method: "round_robin" - - decode: - nodes: 3 - workers: 3 - env: - SGLANG_ENABLE_SPEC_V2: "1" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 10 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.82 - max-running-requests: 32 - cuda-graph-max-bs: 32 - - # MTP settings - speculative-algorithm: "EAGLE" - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "4x8x16x32x64" - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs16-1p3d-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs16-1p3d-stp.yaml deleted file mode 100644 index 6562d0fa67..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs16-1p3d-stp.yaml +++ /dev/null @@ -1,115 +0,0 @@ -schema: 2 -name: "bs16-1p3d-h200-fp8" - -model: - path: "dsr1" - container: "lmsysorg/sglang:v0.5.8.post1-cu130" - precision: "fp8" - -frontend: - nginx_container: nginx - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - # stream-interval: 50 - watchdog-timeout: 1000000 - max-running-requests: 16 - - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.82 - max-prefill-tokens: 32768 - chunked-prefill-size: 32768 - - # Request handling - load-balance-method: "round_robin" - - decode: - nodes: 3 - workers: 3 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 10 - watchdog-timeout: 1000000 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.82 - max-running-requests: 32 - cuda-graph-max-bs: 32 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "8x16x32" - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs4-1p7d-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs4-1p7d-mtp.yaml deleted file mode 100644 index a7a5d2c335..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs4-1p7d-mtp.yaml +++ /dev/null @@ -1,124 +0,0 @@ -schema: 2 -name: "bs4-1p7d-h200-fp8-mtp" - -model: - path: "dsr1" - container: "lmsysorg/sglang:v0.5.8.post1-cu130" - precision: "fp8" - -frontend: - nginx_container: nginx - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - SGLANG_ENABLE_SPEC_V2: "1" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - # stream-interval: 50 - max-running-requests: 16 - - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.82 - max-prefill-tokens: 32768 - chunked-prefill-size: 32768 - - # Request handling - load-balance-method: "round_robin" - - decode: - nodes: 7 - workers: 7 - env: - SGLANG_ENABLE_SPEC_V2: "1" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 10 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.75 - max-running-requests: 2 - cuda-graph-max-bs: 2 - - # MTP settings - speculative-algorithm: "EAGLE" - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1x4x8" - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs4-1p7d-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs4-1p7d-stp.yaml deleted file mode 100644 index 78807ab23d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs4-1p7d-stp.yaml +++ /dev/null @@ -1,115 +0,0 @@ -schema: 2 -name: "bs4-1p7d-h200-fp8" - -model: - path: "dsr1" - container: "lmsysorg/sglang:v0.5.8.post1-cu130" - precision: "fp8" - -frontend: - nginx_container: nginx - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - # stream-interval: 50 - watchdog-timeout: 1000000 - max-running-requests: 16 - - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.82 - max-prefill-tokens: 32768 - chunked-prefill-size: 32768 - - # Request handling - load-balance-method: "round_robin" - - decode: - nodes: 7 - workers: 7 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 10 - watchdog-timeout: 1000000 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.82 - max-running-requests: 8 - cuda-graph-max-bs: 8 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1x4x8" - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs64-2p3d-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs64-2p3d-mtp.yaml deleted file mode 100644 index e356fa8cb0..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs64-2p3d-mtp.yaml +++ /dev/null @@ -1,133 +0,0 @@ -schema: 2 -name: "bs64-2p3d-h200-fp8-mtp" - -model: - path: "dsr1" - container: "lmsysorg/sglang:v0.5.8.post1-cu130" - precision: "fp8" - -frontend: - nginx_container: nginx - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -engine: sglang -roles: - prefill: - nodes: 2 - workers: 2 - env: - SGLANG_ENABLE_SPEC_V2: "1" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - # stream-interval: 50 - max-running-requests: 16 - - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.82 - max-prefill-tokens: 32768 - chunked-prefill-size: 32768 - - # Request handling - load-balance-method: "round_robin" - - decode: - nodes: 3 - workers: 3 - env: - SGLANG_ENABLE_SPEC_V2: "1" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 10 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - context-length: 72000 - max-total-tokens: 128000 - # Memory and token limits - mem-fraction-static: 0.75 - max-running-requests: 16 - cuda-graph-max-bs: 16 - - # MTP settings - speculative-algorithm: "EAGLE" - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "32x64x128" - req_rate: "inf" - -# benchmark: -# type: "gpqa" -# num_examples: 198 -# repeat: 4 -# num_threads: 32 -# max_tokens: 64000 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs64-2p3d-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs64-2p3d-stp.yaml deleted file mode 100644 index 781d346906..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs64-2p3d-stp.yaml +++ /dev/null @@ -1,124 +0,0 @@ -schema: 2 -name: "bs64-2p3d-h200-fp8" - -model: - path: "dsr1" - container: "lmsysorg/sglang:v0.5.8.post1-cu130" - precision: "fp8" - -frontend: - nginx_container: nginx - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -engine: sglang -roles: - prefill: - nodes: 2 - workers: 2 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - # stream-interval: 50 - watchdog-timeout: 1000000 - max-running-requests: 16 - - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.82 - max-prefill-tokens: 32768 - chunked-prefill-size: 32768 - - # Request handling - load-balance-method: "round_robin" - - decode: - nodes: 3 - workers: 3 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 10 - watchdog-timeout: 1000000 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - #context-length: 72000 - # max-total-tokens: 128000 - # Memory and token limits - mem-fraction-static: 0.82 - max-running-requests: 128 - cuda-graph-max-bs: 128 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "32x64x128" - req_rate: "inf" - -# benchmark: -# type: "gpqa" -# num_examples: 198 -# repeat: 4 -# num_threads: 32 -# max_tokens: 64000 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs8-1p6d-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs8-1p6d-mtp.yaml deleted file mode 100644 index 17c0d6e297..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs8-1p6d-mtp.yaml +++ /dev/null @@ -1,125 +0,0 @@ -schema: 2 -name: "bs8-1p6d-h200-fp8-mtp" - -model: - path: "dsr1" - container: "lmsysorg/sglang:v0.5.8.post1-cu130" - precision: "fp8" - -frontend: - nginx_container: nginx - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - SGLANG_ENABLE_SPEC_V2: "1" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - # Decode-specific environment variables - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - # stream-interval: 50 - max-running-requests: 16 - - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.82 - max-prefill-tokens: 32768 - chunked-prefill-size: 32768 - - # Request handling - load-balance-method: "round_robin" - - - decode: - nodes: 6 - workers: 6 - env: - SGLANG_ENABLE_SPEC_V2: "1" - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - watchdog-timeout: 1000000 - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 10 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.82 - max-running-requests: 16 - cuda-graph-max-bs: 16 - - # MTP settings - speculative-algorithm: "EAGLE" - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "2x4x8x16x32" - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs8-1p6d-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs8-1p6d-stp.yaml deleted file mode 100644 index 11afe7a270..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs8-1p6d-stp.yaml +++ /dev/null @@ -1,116 +0,0 @@ -schema: 2 -name: "bs8-1p6d-h200-fp8" - -model: - path: "dsr1" - container: "lmsysorg/sglang:v0.5.8.post1-cu130" - precision: "fp8" - -frontend: - nginx_container: nginx - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -dynamo: - # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" - - source: - pypi: "0.8.0" -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Radix cache disabled - disable-radix-cache: true - - # Other flags - # stream-interval: 50 - watchdog-timeout: 1000000 - max-running-requests: 16 - - - # Prefill-specific mode - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.82 - max-prefill-tokens: 32768 - chunked-prefill-size: 32768 - - # Request handling - load-balance-method: "round_robin" - - - decode: - nodes: 6 - workers: 6 - env: - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - - args: - # Model configuration - served-model-name: "deepseek-ai/DeepSeek-R1" - model-path: "/model/" - skip-tokenizer-init: true - trust-remote-code: true - - # Parallelism - tp-size: 8 - dp-size: 1 - ep-size: 1 - - # KV cache and attention - attention-backend: "flashinfer" - - # Other flags - disable-radix-cache: true - stream-interval: 10 - watchdog-timeout: 1000000 - - # Disagg - disaggregation-bootstrap-port: 30001 - disaggregation-mode: "decode" - disaggregation-transfer-backend: nixl - - # Memory and token limits - mem-fraction-static: 0.82 - max-running-requests: 16 - cuda-graph-max-bs: 16 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "4x8x16" - req_rate: "inf" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml new file mode 100644 index 0000000000..cd6d508a73 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml @@ -0,0 +1,391 @@ +# srt-slurm recipes for dsr1/sglang/h200-fp8/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_. + +schema: 2 + +base: + model: + path: dsr1 + container: lmsysorg/sglang:v0.5.8.post1-cu130 + precision: fp8 + frontend: + nginx_container: nginx + resources: + gpu_type: h200 + gpus_per_node: 8 + dynamo: + # Dynamo 0.8.0 was the default when this recipe was written; pin it explicitly. + # Use NATS for a recipe prior to Dynamo commit 39d2a68. + request_plane: nats + source: + pypi: 0.8.0 + engine: sglang + roles: + prefill: + env: + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + # Memory and token limits + args: + # Model configuration + served-model-name: deepseek-ai/DeepSeek-R1 + model-path: /model/ + skip-tokenizer-init: true + trust-remote-code: true + watchdog-timeout: 1000000 + # Parallelism + tp-size: 8 + dp-size: 1 + ep-size: 1 + # KV cache and attention + attention-backend: flashinfer + # Radix cache disabled + disable-radix-cache: true + max-running-requests: 16 + # Prefill-specific mode + disaggregation-bootstrap-port: 30001 + disaggregation-mode: prefill + disaggregation-transfer-backend: nixl + # Request handling + load-balance-method: round_robin + decode: + env: + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + # Memory and token limits + args: + # Model configuration + served-model-name: deepseek-ai/DeepSeek-R1 + model-path: /model/ + skip-tokenizer-init: true + trust-remote-code: true + watchdog-timeout: 1000000 + # Parallelism + tp-size: 8 + # KV cache and attention + attention-backend: flashinfer + # Other flags + disable-radix-cache: true + stream-interval: 10 + # Disagg + disaggregation-bootstrap-port: 30001 + disaggregation-mode: decode + disaggregation-transfer-backend: nixl + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + +override_disagg_bs128_1p1d_dep_mtp: + name: bs128-1p1d-dep-h200-fp8-mtp + roles: + prefill: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + # Decode-specific environment variables + # Other flags + # stream-interval: 50 + args: + mem-fraction-static: 0.75 + max-prefill-tokens: 163840 + chunked-prefill-size: 163840 + decode: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + mem-fraction-static: 0.85 + max-running-requests: 192 + cuda-graph-max-bs: 192 + # MTP settings + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + benchmark: + concurrencies: 32x64x128x256x512 + +override_disagg_bs128_1p1d_dep_stp: + name: bs128-1p1d-dep-h200-fp8 + roles: + prefill: + nodes: 1 + workers: 1 + # Other flags + # stream-interval: 50 + args: + mem-fraction-static: 0.75 + max-prefill-tokens: 163840 + chunked-prefill-size: 163840 + decode: + nodes: 1 + workers: 1 + args: + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + mem-fraction-static: 0.88 + max-running-requests: 256 + cuda-graph-max-bs: 256 + benchmark: + concurrencies: 64x128x256 + +override_disagg_bs16_1p3d_mtp: + name: bs16-1p3d-h200-fp8-mtp + roles: + prefill: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + # Decode-specific environment variables + # Other flags + # stream-interval: 50 + args: + mem-fraction-static: 0.82 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + decode: + nodes: 3 + workers: 3 + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + dp-size: 1 + ep-size: 1 + mem-fraction-static: 0.82 + max-running-requests: 32 + cuda-graph-max-bs: 32 + # MTP settings + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + benchmark: + concurrencies: 4x8x16x32x64 + +override_disagg_bs16_1p3d_stp: + name: bs16-1p3d-h200-fp8 + roles: + prefill: + nodes: 1 + workers: 1 + # Other flags + # stream-interval: 50 + args: + mem-fraction-static: 0.82 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + decode: + nodes: 3 + workers: 3 + args: + dp-size: 1 + ep-size: 1 + mem-fraction-static: 0.82 + max-running-requests: 32 + cuda-graph-max-bs: 32 + benchmark: + concurrencies: 8x16x32 + +override_disagg_bs4_1p7d_mtp: + name: bs4-1p7d-h200-fp8-mtp + roles: + prefill: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + # Decode-specific environment variables + # Other flags + # stream-interval: 50 + args: + mem-fraction-static: 0.82 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + decode: + nodes: 7 + workers: 7 + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + dp-size: 1 + ep-size: 1 + mem-fraction-static: 0.75 + max-running-requests: 2 + cuda-graph-max-bs: 2 + # MTP settings + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + benchmark: + concurrencies: 1x4x8 + +override_disagg_bs4_1p7d_stp: + name: bs4-1p7d-h200-fp8 + roles: + prefill: + nodes: 1 + workers: 1 + # Other flags + # stream-interval: 50 + args: + mem-fraction-static: 0.82 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + decode: + nodes: 7 + workers: 7 + args: + dp-size: 1 + ep-size: 1 + mem-fraction-static: 0.82 + max-running-requests: 8 + cuda-graph-max-bs: 8 + benchmark: + concurrencies: 1x4x8 + +# benchmark: +# type: "gpqa" +# num_examples: 198 +# repeat: 4 +# num_threads: 32 +# max_tokens: 64000 +override_disagg_bs64_2p3d_mtp: + name: bs64-2p3d-h200-fp8-mtp + roles: + prefill: + nodes: 2 + workers: 2 + env: + SGLANG_ENABLE_SPEC_V2: '1' + # Decode-specific environment variables + # Other flags + # stream-interval: 50 + args: + mem-fraction-static: 0.82 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + decode: + nodes: 3 + workers: 3 + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + dp-size: 1 + ep-size: 1 + mem-fraction-static: 0.75 + max-running-requests: 16 + cuda-graph-max-bs: 16 + # MTP settings + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + context-length: 72000 + max-total-tokens: 128000 + benchmark: + concurrencies: 32x64x128 + +# benchmark: +# type: "gpqa" +# num_examples: 198 +# repeat: 4 +# num_threads: 32 +# max_tokens: 64000 +override_disagg_bs64_2p3d_stp: + name: bs64-2p3d-h200-fp8 + roles: + prefill: + nodes: 2 + workers: 2 + # Other flags + # stream-interval: 50 + args: + mem-fraction-static: 0.82 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + decode: + nodes: 3 + workers: 3 + args: + dp-size: 1 + ep-size: 1 + #context-length: 72000 + # max-total-tokens: 128000 + mem-fraction-static: 0.82 + max-running-requests: 128 + cuda-graph-max-bs: 128 + benchmark: + concurrencies: 32x64x128 + +override_disagg_bs8_1p6d_mtp: + name: bs8-1p6d-h200-fp8-mtp + roles: + prefill: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + # Decode-specific environment variables + # Other flags + # stream-interval: 50 + args: + mem-fraction-static: 0.82 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + decode: + nodes: 6 + workers: 6 + env: + SGLANG_ENABLE_SPEC_V2: '1' + args: + dp-size: 1 + ep-size: 1 + mem-fraction-static: 0.82 + max-running-requests: 16 + cuda-graph-max-bs: 16 + # MTP settings + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + benchmark: + concurrencies: 2x4x8x16x32 + +override_disagg_bs8_1p6d_stp: + name: bs8-1p6d-h200-fp8 + roles: + prefill: + nodes: 1 + workers: 1 + # Other flags + # stream-interval: 50 + args: + mem-fraction-static: 0.82 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + decode: + nodes: 6 + workers: 6 + args: + dp-size: 1 + ep-size: 1 + mem-fraction-static: 0.82 + max-running-requests: 16 + cuda-graph-max-bs: 16 + benchmark: + concurrencies: 4x8x16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p1d-dep8-b8-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p1d-dep8-b8-eplb0-mtp3.yaml deleted file mode 100644 index f24a9de07f..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p1d-dep8-b8-eplb0-mtp3.yaml +++ /dev/null @@ -1,120 +0,0 @@ -schema: 2 -name: "ctx1_gen1_dep8_batch8_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 1 - - workers: 1 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 8 - max_num_tokens: 32 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 5 - - 6 - - 7 - - 8 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "90" - req_rate: "inf" - -frontend: - nginx_container: "nginx-sqsh" - type: "dynamo" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp3.yaml deleted file mode 100644 index 40cdb3c74b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp3.yaml +++ /dev/null @@ -1,127 +0,0 @@ -schema: 2 -name: "ctx1_gen3_tep8_batch16_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 3 - - workers: 3 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 16 - max_num_tokens: 64 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 9 - - 10 - - 11 - - 12 - - 13 - - 14 - - 15 - - 16 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "66" - req_rate: "inf" - -frontend: - nginx_container: "nginx-sqsh" - type: "dynamo" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b1-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b1-eplb0-mtp0.yaml deleted file mode 100644 index 4819f17f8e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b1-eplb0-mtp0.yaml +++ /dev/null @@ -1,111 +0,0 @@ -schema: 2 -name: "ctx1_gen5_tep8_batch1_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - - decode: - nodes: 5 - - workers: 5 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 1 - max_num_tokens: 1 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "6" - req_rate: "inf" - -frontend: - nginx_container: "nginx-sqsh" - type: "dynamo" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b1-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b1-eplb0-mtp3.yaml deleted file mode 100644 index 8c46f69689..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b1-eplb0-mtp3.yaml +++ /dev/null @@ -1,116 +0,0 @@ -schema: 2 -name: "ctx1_gen5_tep8_batch1_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 5 - - workers: 5 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 1 - max_num_tokens: 4 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "6" - req_rate: "inf" - -frontend: - nginx_container: "nginx-sqsh" - type: "dynamo" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b16-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b16-eplb0-mtp0.yaml deleted file mode 100644 index 9d01d61c99..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b16-eplb0-mtp0.yaml +++ /dev/null @@ -1,124 +0,0 @@ -schema: 2 -name: "ctx1_gen5_tep8_batch8_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - - decode: - nodes: 5 - - workers: 5 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 16 - max_num_tokens: 16 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 3 - - 4 - - 5 - - 6 - - 7 - - 8 - - 9 - - 10 - - 12 - - 13 - - 14 - - 15 - - 16 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "10x15x25x50x100" - req_rate: "inf" - -frontend: - nginx_container: "nginx-sqsh" - type: "dynamo" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b8-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b8-eplb0-mtp3.yaml deleted file mode 100644 index 05dd640962..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b8-eplb0-mtp3.yaml +++ /dev/null @@ -1,123 +0,0 @@ -schema: 2 -name: "ctx1_gen5_tep8_batch8_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 5 - - workers: 5 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 8 - max_num_tokens: 32 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 3 - - 4 - - 5 - - 6 - - 7 - - 8 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "10x15x30x60" - req_rate: "inf" - -frontend: - nginx_container: "nginx-sqsh" - type: "dynamo" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-2p5d-tep8-b64-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-2p5d-tep8-b64-eplb0-mtp0.yaml deleted file mode 100644 index ccdabf7309..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-2p5d-tep8-b64-eplb0-mtp0.yaml +++ /dev/null @@ -1,119 +0,0 @@ -schema: 2 -name: "ctx2_gen5_tep8_batch64_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 2 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - - decode: - nodes: 5 - - workers: 5 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 64 - max_num_tokens: 64 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 58 - - 60 - - 62 - - 64 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "370" - req_rate: "inf" - -frontend: - nginx_container: "nginx-sqsh" - type: "dynamo" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp3.yaml deleted file mode 100644 index 5029cd5517..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp3.yaml +++ /dev/null @@ -1,124 +0,0 @@ -schema: 2 -name: "ctx3_gen1_dep8_batch64_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 3 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 1 - - workers: 1 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 64 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 48 - - 56 - - 60 - - 62 - - 64 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "548" - req_rate: "inf" - -frontend: - nginx_container: "nginx-sqsh" - type: "dynamo" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-4p1d-dep8-b192-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-4p1d-dep8-b192-eplb0-mtp0.yaml deleted file mode 100644 index f153ec6020..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-4p1d-dep8-b192-eplb0-mtp0.yaml +++ /dev/null @@ -1,122 +0,0 @@ -schema: 2 -name: "ctx4_gen1_dep8_batch192_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 4 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - - decode: - nodes: 1 - - workers: 1 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 192 - max_num_tokens: 192 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 152 - - 160 - - 168 - - 176 - - 184 - - 190 - - 192 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1606" - req_rate: "inf" - -frontend: - nginx_container: "nginx-sqsh" - type: "dynamo" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-4p3d-dep8-b32-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-4p3d-dep8-b32-eplb0-mtp0.yaml deleted file mode 100644 index def3f8dbab..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-4p3d-dep8-b32-eplb0-mtp0.yaml +++ /dev/null @@ -1,115 +0,0 @@ -schema: 2 -name: "ctx4_gen3_dep8_batch32_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 4 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - - decode: - nodes: 3 - - workers: 3 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 32 - max_num_tokens: 32 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 28 - - 30 - - 32 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "837" - req_rate: "inf" - -frontend: - nginx_container: "nginx-sqsh" - type: "dynamo" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-5p1d-dep8-b192-eplb0-mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-5p1d-dep8-b192-eplb0-mtp1.yaml deleted file mode 100644 index c4df00f6a7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-5p1d-dep8-b192-eplb0-mtp1.yaml +++ /dev/null @@ -1,128 +0,0 @@ -schema: 2 -name: "ctx5_gen1_dep8_batch192_eplb0_mtp1" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 3 - workers: 5 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - - decode: - nodes: 1 - - workers: 1 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 192 - max_num_tokens: 384 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 130 - - 132 - - 134 - - 136 - - 138 - - 168 - - 192 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1096x1691" - req_rate: "inf" - -frontend: - nginx_container: "nginx-sqsh" - type: "dynamo" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-5p2d-dep8-b32-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-5p2d-dep8-b32-eplb0-mtp3.yaml deleted file mode 100644 index 2cd5e90785..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-5p2d-dep8-b32-eplb0-mtp3.yaml +++ /dev/null @@ -1,123 +0,0 @@ -schema: 2 -name: "ctx5_gen2_dep8_batch32_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 3 - workers: 5 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 2 - - workers: 2 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 20 - - 24 - - 28 - - 30 - - 32 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "658" - req_rate: "inf" - -frontend: - nginx_container: "nginx-sqsh" - type: "dynamo" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-7p2d-dep8-b128-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-7p2d-dep8-b128-eplb0-mtp0.yaml deleted file mode 100644 index 825cee57c7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-7p2d-dep8-b128-eplb0-mtp0.yaml +++ /dev/null @@ -1,118 +0,0 @@ -schema: 2 -name: "ctx7_gen2_dep8_batch128_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 4 - workers: 7 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - - decode: - nodes: 2 - - workers: 2 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_RNDV_SCHEME: "put_zcopy" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 128 - max_num_tokens: 128 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 116 - - 120 - - 124 - - 128 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "2222" - req_rate: "inf" - -frontend: - nginx_container: "nginx-sqsh" - type: "dynamo" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml new file mode 100644 index 0000000000..f8b418b040 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml @@ -0,0 +1,455 @@ +# srt-slurm recipes for dsr1/trtllm/b200-fp4/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_. + +schema: 2 + +base: + model: + path: dsr1 + container: dynamo-trtllm + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + engine: trtllm + roles: + prefill: + gpus: 4 + env: + NCCL_GRAPH_MIXING_SUPPORT: '0' + OMPI_MCA_coll_ucc_enable: '0' + TLLM_ALL_RANK_LOG: '1' + TLLM_LOG_LEVEL: INFO + TRTLLM_ENABLE_PDL: '1' + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + UCX_CUDA_IPC_ENABLE_MNNVL: n + UCX_RNDV_SCHEME: put_zcopy + args: + max_batch_size: 2 + max_num_tokens: 16896 + max_seq_len: 8232 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + moe_config: + backend: TRTLLM + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + decode: + env: + NCCL_GRAPH_MIXING_SUPPORT: '0' + OMPI_MCA_coll_ucc_enable: '0' + TLLM_ALL_RANK_LOG: '1' + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + UCX_CUDA_IPC_ENABLE_MNNVL: n + UCX_RNDV_SCHEME: put_zcopy + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + pipeline_parallel_size: 1 + max_seq_len: 9256 + cuda_graph_config: + enable_padding: true + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + dtype: fp8 + moe_config: + use_low_precision_moe_combine: true + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 100 + num_postprocess_workers: 4 + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + frontend: + nginx_container: nginx-sqsh + type: dynamo + dynamo: + install: false + # The unversioned container alias does not identify a Dynamo version or hash. + # Use NATS for a recipe with an unverified Dynamo revision. + request_plane: nats + +override_disagg_1p1d_dep8_b8_eplb0_mtp3: + name: ctx1_gen1_dep8_batch8_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 1 + workers: 1 + args: + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 8 + max_num_tokens: 32 + cuda_graph_config: + batch_sizes: [1, 2, 4, 5, 6, 7, 8] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '90' + +override_disagg_1p3d_tep8_b16_eplb0_mtp3: + name: ctx1_gen3_tep8_batch16_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 3 + workers: 3 + env: + TRTLLM_ENABLE_PDL: '1' + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 16 + max_num_tokens: 64 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 9, 10, 11, 12, 13, 14, 15, 16] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + benchmark: + concurrencies: '66' + +override_disagg_1p5d_tep8_b1_eplb0_mtp0: + name: ctx1_gen5_tep8_batch1_eplb0_mtp0 + roles: + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 5 + workers: 5 + env: + TRTLLM_ENABLE_PDL: '1' + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 1 + max_num_tokens: 1 + cuda_graph_config: + batch_sizes: [1] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + allreduce_strategy: MNNVL + benchmark: + concurrencies: '6' + +override_disagg_1p5d_tep8_b1_eplb0_mtp3: + name: ctx1_gen5_tep8_batch1_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 5 + workers: 5 + env: + TRTLLM_ENABLE_PDL: '1' + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 1 + max_num_tokens: 4 + cuda_graph_config: + batch_sizes: [1] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + benchmark: + concurrencies: '6' + +override_disagg_1p5d_tep8_b16_eplb0_mtp0: + name: ctx1_gen5_tep8_batch8_eplb0_mtp0 + roles: + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 5 + workers: 5 + env: + TRTLLM_ENABLE_PDL: '1' + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 16 + max_num_tokens: 16 + cuda_graph_config: + batch_sizes: [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 14, 15, 16] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + allreduce_strategy: MNNVL + benchmark: + concurrencies: 10x15x25x50x100 + +override_disagg_1p5d_tep8_b8_eplb0_mtp3: + name: ctx1_gen5_tep8_batch8_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 5 + workers: 5 + env: + TRTLLM_ENABLE_PDL: '1' + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 8 + max_num_tokens: 32 + cuda_graph_config: + batch_sizes: [1, 2, 3, 4, 5, 6, 7, 8] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + benchmark: + concurrencies: 10x15x30x60 + +override_disagg_2p5d_tep8_b64_eplb0_mtp0: + name: ctx2_gen5_tep8_batch64_eplb0_mtp0 + roles: + prefill: + nodes: 1 + workers: 2 + decode: + nodes: 5 + workers: 5 + env: + TRTLLM_ENABLE_PDL: '1' + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 64 + max_num_tokens: 64 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 58, 60, 62, 64] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + allreduce_strategy: MNNVL + benchmark: + concurrencies: '370' + +override_disagg_3p1d_dep8_b64_eplb0_mtp3: + name: ctx3_gen1_dep8_batch64_eplb0_mtp3 + roles: + prefill: + nodes: 2 + workers: 3 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 1 + workers: 1 + args: + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 64 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 48, 56, 60, 62, 64] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '548' + +override_disagg_4p1d_dep8_b192_eplb0_mtp0: + name: ctx4_gen1_dep8_batch192_eplb0_mtp0 + roles: + prefill: + nodes: 2 + workers: 4 + decode: + nodes: 1 + workers: 1 + args: + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 192 + max_num_tokens: 192 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128, 152, 160, 168, 176, 184, 190, 192] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTLASS + benchmark: + concurrencies: '1606' + +override_disagg_4p3d_dep8_b32_eplb0_mtp0: + name: ctx4_gen3_dep8_batch32_eplb0_mtp0 + roles: + prefill: + nodes: 2 + workers: 4 + decode: + nodes: 3 + workers: 3 + args: + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 32 + max_num_tokens: 32 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 28, 30, 32] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTLASS + benchmark: + concurrencies: '837' + +override_disagg_5p1d_dep8_b192_eplb0_mtp1: + name: ctx5_gen1_dep8_batch192_eplb0_mtp1 + roles: + prefill: + nodes: 3 + workers: 5 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + decode: + nodes: 1 + workers: 1 + args: + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 192 + max_num_tokens: 384 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128, 130, 132, 134, 136, 138, 168, 192] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + benchmark: + concurrencies: 1096x1691 + +override_disagg_5p2d_dep8_b32_eplb0_mtp3: + name: ctx5_gen2_dep8_batch32_eplb0_mtp3 + roles: + prefill: + nodes: 3 + workers: 5 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 2 + workers: 2 + args: + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 32 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 20, 24, 28, 30, 32] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '658' + +override_disagg_7p2d_dep8_b128_eplb0_mtp0: + name: ctx7_gen2_dep8_batch128_eplb0_mtp0 + roles: + prefill: + nodes: 4 + workers: 7 + decode: + nodes: 2 + workers: 2 + args: + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 128 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 116, 120, 124, 128] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTLASS + benchmark: + concurrencies: '2222' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp1.yaml deleted file mode 100644 index b770fc7e2b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp1.yaml +++ /dev/null @@ -1,133 +0,0 @@ -schema: 2 -name: "ctx10_gen1_dep8_batch256_eplb0_mtp1" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 3 - workers: 10 - gpus: 2 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - - decode: - nodes: 1 - - workers: 1 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 256 - max_num_tokens: 512 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "2198" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep4-b32-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep4-b32-eplb0-mtp0.yaml deleted file mode 100644 index a944acc858..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep4-b32-eplb0-mtp0.yaml +++ /dev/null @@ -1,127 +0,0 @@ -schema: 2 -name: "ctx1_gen3_tep4_batch32_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - - decode: - nodes: 2 - workers: 3 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - - args: - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 32 - max_num_tokens: 32 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "105" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep8-b1-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep8-b1-eplb0-mtp0.yaml deleted file mode 100644 index 43293b8635..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep8-b1-eplb0-mtp0.yaml +++ /dev/null @@ -1,134 +0,0 @@ -schema: 2 -name: "ctx1_gen3_tep8_batch1_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - - decode: - nodes: 3 - - workers: 3 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 1 - max_num_tokens: 1 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 768 - - 1024 - - 2048 - - 1 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "4" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0.yaml deleted file mode 100644 index 8298e611d9..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0.yaml +++ /dev/null @@ -1,125 +0,0 @@ -schema: 2 -name: "ctx1_gen3_tep8_batch16_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - - decode: - nodes: 3 - - workers: 3 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 16 - max_num_tokens: 16 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "63" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep4-b2-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep4-b2-eplb0-mtp0.yaml deleted file mode 100644 index c468b552c8..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep4-b2-eplb0-mtp0.yaml +++ /dev/null @@ -1,122 +0,0 @@ -schema: 2 -name: "ctx1_gen4_tep4_batch2_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - - decode: - nodes: 2 - - workers: 4 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - - args: - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 2 - max_num_tokens: 2 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "12" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep4-b8-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep4-b8-eplb0-mtp3.yaml deleted file mode 100644 index 6e6e3f4339..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep4-b8-eplb0-mtp3.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: "ctx1_gen4_tep4_batch8_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 2 - - workers: 4 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - - args: - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 8 - max_num_tokens: 32 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - gpus_per_node: 4 - allreduce_strategy: MNNVL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "52" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3.yaml deleted file mode 100644 index fdadfd7af9..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3.yaml +++ /dev/null @@ -1,127 +0,0 @@ -schema: 2 -name: "ctx1_gen4_tep8_batch1_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 4 - - workers: 4 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 1 - max_num_tokens: 4 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "8" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3.yaml deleted file mode 100644 index 6ebfa26856..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3.yaml +++ /dev/null @@ -1,130 +0,0 @@ -schema: 2 -name: "ctx1_gen4_tep8_batch4_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 4 - - workers: 4 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 4 - max_num_tokens: 16 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 3 - - 4 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "32" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-3p1d-dep8-b16-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-3p1d-dep8-b16-eplb0-mtp3.yaml deleted file mode 100644 index 17806d7cd7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-3p1d-dep8-b16-eplb0-mtp3.yaml +++ /dev/null @@ -1,129 +0,0 @@ -schema: 2 -name: "ctx3_gen1_dep8_batch16_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 3 - gpus: 2 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 1 - - workers: 1 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 16 - max_num_tokens: 64 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "181" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-5p2d-dep8-b32-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-5p2d-dep8-b32-eplb0-mtp0.yaml deleted file mode 100644 index def8f83d7d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-5p2d-dep8-b32-eplb0-mtp0.yaml +++ /dev/null @@ -1,124 +0,0 @@ -schema: 2 -name: "ctx5_gen2_dep8_batch32_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 5 - gpus: 2 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - - decode: - nodes: 2 - - workers: 2 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 32 - max_num_tokens: 32 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "589" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-6p1d-dep8-b128-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-6p1d-dep8-b128-eplb0-mtp0.yaml deleted file mode 100644 index d9432e68c0..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-6p1d-dep8-b128-eplb0-mtp0.yaml +++ /dev/null @@ -1,132 +0,0 @@ -schema: 2 -name: "ctx6_gen1_dep8_batch128_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 6 - gpus: 2 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - - decode: - nodes: 1 - - workers: 1 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 128 - max_num_tokens: 128 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 768 - - 1024 - - 2048 - - 128 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1093" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-8p1d-dep8-b256-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-8p1d-dep8-b256-eplb0-mtp0.yaml deleted file mode 100644 index f4cde2cac2..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-8p1d-dep8-b256-eplb0-mtp0.yaml +++ /dev/null @@ -1,132 +0,0 @@ -schema: 2 -name: "ctx8_gen1_dep8_batch256_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 8 - gpus: 2 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - - decode: - nodes: 1 - - workers: 1 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 256 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 768 - - 1024 - - 2048 - - 256 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "2048" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-9p1d-dep8-b128-eplb0-mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-9p1d-dep8-b128-eplb0-mtp1.yaml deleted file mode 100644 index 7ee89a73a8..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-9p1d-dep8-b128-eplb0-mtp1.yaml +++ /dev/null @@ -1,132 +0,0 @@ -schema: 2 -name: "ctx9_gen1_dep8_batch128_eplb0_mtp1" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 3 - workers: 9 - gpus: 2 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - - decode: - nodes: 1 - - workers: 1 - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 128 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8448 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1197" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml new file mode 100644 index 0000000000..8a9e1c72e9 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml @@ -0,0 +1,487 @@ +# srt-slurm recipes for dsr1/trtllm/b300-fp4/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_. + +schema: 2 + +base: + model: + path: dsr1 + container: dynamo-trtllm + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + engine: trtllm + roles: + prefill: + gpus: 2 + env: + NCCL_GRAPH_MIXING_SUPPORT: '0' + OMPI_MCA_coll_ucc_enable: '0' + TLLM_ALL_RANK_LOG: '1' + TLLM_LOG_LEVEL: INFO + TRTLLM_ENABLE_PDL: '1' + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + UCX_CUDA_IPC_ENABLE_MNNVL: n + UCX_MAX_RMA_RAILS: '1' + UCX_MAX_RNDV_RAILS: '1' + UCX_RNDV_SCHEME: put_zcopy + OMPI_MCA_btl: tcp,self + OMPI_MCA_pml: ob1 + TRTLLM_UCX_INTERFACE: mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1 + args: + max_batch_size: 2 + max_num_tokens: 16896 + max_seq_len: 8232 + tensor_parallel_size: 2 + moe_expert_parallel_size: 2 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + moe_config: + backend: TRTLLM + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.75 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + decode: + env: + NCCL_GRAPH_MIXING_SUPPORT: '0' + OMPI_MCA_coll_ucc_enable: '0' + TLLM_ALL_RANK_LOG: '1' + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + UCX_CUDA_IPC_ENABLE_MNNVL: n + UCX_MAX_RMA_RAILS: '1' + UCX_MAX_RNDV_RAILS: '1' + UCX_RNDV_SCHEME: put_zcopy + OMPI_MCA_btl: tcp,self + OMPI_MCA_pml: ob1 + TRTLLM_UCX_INTERFACE: mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1 + args: + pipeline_parallel_size: 1 + max_seq_len: 9256 + cuda_graph_config: + enable_padding: true + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + dtype: fp8 + moe_config: + use_low_precision_moe_combine: true + cache_transceiver_config: + max_tokens_in_buffer: 8448 + backend: UCX + stream_interval: 100 + num_postprocess_workers: 4 + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + frontend: + type: dynamo + enable_multiple_frontends: false + dynamo: + install: false + # The unversioned container alias does not identify a Dynamo version or hash. + # Use NATS for a recipe with an unverified Dynamo revision. + request_plane: nats + +override_disagg_10p1d_dep8_b256_eplb0_mtp1: + name: ctx10_gen1_dep8_batch256_eplb0_mtp1 + roles: + prefill: + nodes: 3 + workers: 10 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + decode: + nodes: 1 + workers: 1 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 256 + max_num_tokens: 512 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128, 256] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + benchmark: + concurrencies: '2198' + +override_disagg_1p3d_tep4_b32_eplb0_mtp0: + name: ctx1_gen3_tep4_batch32_eplb0_mtp0 + roles: + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 2 + workers: 3 + env: + TRTLLM_ENABLE_PDL: '1' + args: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 32 + max_num_tokens: 32 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + allreduce_strategy: MNNVL + gpus: 4 + benchmark: + concurrencies: '105' + +override_disagg_1p3d_tep8_b1_eplb0_mtp0: + name: ctx1_gen3_tep8_batch1_eplb0_mtp0 + roles: + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 3 + workers: 3 + env: + TRTLLM_ENABLE_PDL: '1' + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 1 + max_num_tokens: 1 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 768, 1024, 2048, 1] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + allreduce_strategy: MNNVL + benchmark: + concurrencies: '4' + +override_disagg_1p3d_tep8_b16_eplb0_mtp0: + name: ctx1_gen3_tep8_batch16_eplb0_mtp0 + roles: + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 3 + workers: 3 + env: + TRTLLM_ENABLE_PDL: '1' + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 16 + max_num_tokens: 16 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + allreduce_strategy: MNNVL + benchmark: + concurrencies: '63' + +override_disagg_1p4d_tep4_b2_eplb0_mtp0: + name: ctx1_gen4_tep4_batch2_eplb0_mtp0 + roles: + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 2 + workers: 4 + env: + TRTLLM_ENABLE_PDL: '1' + args: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 2 + max_num_tokens: 2 + cuda_graph_config: + batch_sizes: [1, 2] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + allreduce_strategy: MNNVL + benchmark: + concurrencies: '12' + +override_disagg_1p4d_tep4_b8_eplb0_mtp3: + name: ctx1_gen4_tep4_batch8_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 2 + workers: 4 + env: + TRTLLM_ENABLE_PDL: '1' + args: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 8 + max_num_tokens: 32 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + gpus_per_node: 4 + benchmark: + concurrencies: '52' + +override_disagg_1p4d_tep8_b1_eplb0_mtp3: + name: ctx1_gen4_tep8_batch1_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 4 + workers: 4 + env: + TRTLLM_ENABLE_PDL: '1' + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 1 + max_num_tokens: 4 + cuda_graph_config: + batch_sizes: [1] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + benchmark: + concurrencies: '8' + +override_disagg_1p4d_tep8_b4_eplb0_mtp3: + name: ctx1_gen4_tep8_batch4_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 4 + workers: 4 + env: + TRTLLM_ENABLE_PDL: '1' + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 4 + max_num_tokens: 16 + cuda_graph_config: + batch_sizes: [1, 2, 3, 4] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + benchmark: + concurrencies: '32' + +override_disagg_3p1d_dep8_b16_eplb0_mtp3: + name: ctx3_gen1_dep8_batch16_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 3 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 1 + workers: 1 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 16 + max_num_tokens: 64 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '181' + +override_disagg_5p2d_dep8_b32_eplb0_mtp0: + name: ctx5_gen2_dep8_batch32_eplb0_mtp0 + roles: + prefill: + nodes: 2 + workers: 5 + decode: + nodes: 2 + workers: 2 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 32 + max_num_tokens: 32 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTLASS + benchmark: + concurrencies: '589' + +override_disagg_6p1d_dep8_b128_eplb0_mtp0: + name: ctx6_gen1_dep8_batch128_eplb0_mtp0 + roles: + prefill: + nodes: 2 + workers: 6 + decode: + nodes: 1 + workers: 1 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 128 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 768, 1024, 2048, 128] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTLASS + benchmark: + concurrencies: '1093' + +override_disagg_8p1d_dep8_b256_eplb0_mtp0: + name: ctx8_gen1_dep8_batch256_eplb0_mtp0 + roles: + prefill: + nodes: 2 + workers: 8 + decode: + nodes: 1 + workers: 1 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 256 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 768, 1024, 2048, 256] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTLASS + benchmark: + concurrencies: '2048' + +override_disagg_9p1d_dep8_b128_eplb0_mtp1: + name: ctx9_gen1_dep8_batch128_eplb0_mtp1 + roles: + prefill: + nodes: 3 + workers: 9 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + decode: + nodes: 1 + workers: 1 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 128 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + benchmark: + concurrencies: '1197' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p1d-dp8-b8-eplb0-mtp3-c72.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p1d-dp8-b8-eplb0-mtp3-c72.yaml deleted file mode 100644 index a2ce5bd5bf..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p1d-dp8-b8-eplb0-mtp3-c72.yaml +++ /dev/null @@ -1,137 +0,0 @@ -schema: 2 -name: ctx1_gen1_dp8_batch8_eplb0_mtp3_72 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: false - disable_overlap_scheduler: true - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - max_batch_size: 8 - max_num_tokens: 8320 - max_seq_len: 8320 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 4 - - - decode: - nodes: 1 - workers: 1 - gpus: 8 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: true - max_batch_size: 8 - disable_overlap_scheduler: false - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 8 - max_num_tokens: 90 - max_seq_len: 9344 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 20 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [72] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p2d-tp8-b16-eplb0-mtp3-c40.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p2d-tp8-b16-eplb0-mtp3-c40.yaml deleted file mode 100644 index 878b0e29da..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p2d-tp8-b16-eplb0-mtp3-c40.yaml +++ /dev/null @@ -1,138 +0,0 @@ -schema: 2 -name: ctx1_gen2_tp8_batch16_eplb0_mtp3_40 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: false - disable_overlap_scheduler: true - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - max_batch_size: 8 - max_num_tokens: 8320 - max_seq_len: 8320 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 4 - - - decode: - nodes: 2 - workers: 2 - gpus: 8 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: true - max_batch_size: 16 - disable_overlap_scheduler: false - enable_attention_dp: false - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 16 - max_num_tokens: 80 - max_seq_len: 9344 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 20 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [40] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b1-eplb0-mtp3-c8.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b1-eplb0-mtp3-c8.yaml deleted file mode 100644 index 5e6f22c9d6..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b1-eplb0-mtp3-c8.yaml +++ /dev/null @@ -1,138 +0,0 @@ -schema: 2 -name: ctx1_gen4_tp8_batch1_eplb0_mtp3_8 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: false - disable_overlap_scheduler: true - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - max_batch_size: 8 - max_num_tokens: 8320 - max_seq_len: 8320 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 4 - - - decode: - nodes: 4 - workers: 4 - gpus: 8 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: true - max_batch_size: 1 - disable_overlap_scheduler: false - enable_attention_dp: false - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 1 - max_num_tokens: 4 - max_seq_len: 9344 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 20 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [5] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b16-eplb0-mtp0-c64.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b16-eplb0-mtp0-c64.yaml deleted file mode 100644 index 268e130912..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b16-eplb0-mtp0-c64.yaml +++ /dev/null @@ -1,132 +0,0 @@ -schema: 2 -name: ctx1_gen4_tp8_batch16_eplb0_mtp0_64 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: false - disable_overlap_scheduler: true - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - max_batch_size: 8 - max_num_tokens: 8320 - max_seq_len: 8320 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 4 - workers: 4 - gpus: 8 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: true - max_batch_size: 16 - disable_overlap_scheduler: false - enable_attention_dp: false - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 16 - max_num_tokens: 512 - max_seq_len: 9344 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 20 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [64] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b4-eplb0-mtp3-c20.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b4-eplb0-mtp3-c20.yaml deleted file mode 100644 index 7e8eb18108..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b4-eplb0-mtp3-c20.yaml +++ /dev/null @@ -1,138 +0,0 @@ -schema: 2 -name: ctx1_gen4_tp8_batch4_eplb0_mtp3_20 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: false - disable_overlap_scheduler: true - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - max_batch_size: 8 - max_num_tokens: 8320 - max_seq_len: 8320 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 4 - - - decode: - nodes: 4 - workers: 4 - gpus: 8 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: true - max_batch_size: 4 - disable_overlap_scheduler: false - enable_attention_dp: false - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 4 - max_num_tokens: 20 - max_seq_len: 9344 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 20 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [20] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p8d-tp8-b1-eplb0-mtp0-c16.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p8d-tp8-b1-eplb0-mtp0-c16.yaml deleted file mode 100644 index 6ae0592cf4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p8d-tp8-b1-eplb0-mtp0-c16.yaml +++ /dev/null @@ -1,132 +0,0 @@ -schema: 2 -name: ctx1_gen8_tp8_batch2_eplb0_mtp0_16 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: false - disable_overlap_scheduler: true - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - max_batch_size: 8 - max_num_tokens: 8320 - max_seq_len: 8320 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 8 - workers: 8 - gpus: 8 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: true - max_batch_size: 1 - disable_overlap_scheduler: false - enable_attention_dp: false - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 1 - max_num_tokens: 1 - max_seq_len: 9344 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 20 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [10] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-2p1d-dp8-b16-eplb0-mtp3-c144.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-2p1d-dp8-b16-eplb0-mtp3-c144.yaml deleted file mode 100644 index 1973754c90..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-2p1d-dp8-b16-eplb0-mtp3-c144.yaml +++ /dev/null @@ -1,137 +0,0 @@ -schema: 2 -name: ctx2_gen1_dp8_batch16_eplb0_mtp3_144 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 2 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: false - disable_overlap_scheduler: true - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - max_batch_size: 8 - max_num_tokens: 8320 - max_seq_len: 8320 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 4 - - - decode: - nodes: 1 - workers: 1 - gpus: 8 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: true - max_batch_size: 16 - disable_overlap_scheduler: false - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 16 - max_num_tokens: 180 - max_seq_len: 9344 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 20 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [144] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-2p1d-dp8-b32-eplb0-mtp0-c256.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-2p1d-dp8-b32-eplb0-mtp0-c256.yaml deleted file mode 100644 index adc7a95d4a..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-2p1d-dp8-b32-eplb0-mtp0-c256.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: ctx2_gen1_dp8_batch32_eplb0_mtp0_256 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 2 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: false - disable_overlap_scheduler: true - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - max_batch_size: 8 - max_num_tokens: 8320 - max_seq_len: 8320 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 1 - workers: 1 - gpus: 8 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: true - max_batch_size: 32 - disable_overlap_scheduler: false - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 32 - max_num_tokens: 512 - max_seq_len: 9344 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 20 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [256] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-3p1d-dp8-b64-eplb0-mtp0-c512.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-3p1d-dp8-b64-eplb0-mtp0-c512.yaml deleted file mode 100644 index 6185f89f18..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-3p1d-dp8-b64-eplb0-mtp0-c512.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: ctx3_gen1_dp8_batch64_eplb0_mtp0_512 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 3 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: false - disable_overlap_scheduler: true - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - max_batch_size: 8 - max_num_tokens: 8320 - max_seq_len: 8320 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 1 - workers: 1 - gpus: 8 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: true - max_batch_size: 64 - disable_overlap_scheduler: false - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 64 - max_num_tokens: 512 - max_seq_len: 9344 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 20 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [512] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-3p5d-tp8-b64-eplb0-mtp0-c256.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-3p5d-tp8-b64-eplb0-mtp0-c256.yaml deleted file mode 100644 index 9fa54f9385..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-3p5d-tp8-b64-eplb0-mtp0-c256.yaml +++ /dev/null @@ -1,132 +0,0 @@ -schema: 2 -name: ctx3_gen5_tp8_batch64_eplb0_mtp0_256 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 3 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: false - disable_overlap_scheduler: true - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - max_batch_size: 8 - max_num_tokens: 8320 - max_seq_len: 8320 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 5 - workers: 5 - gpus: 8 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: true - max_batch_size: 64 - disable_overlap_scheduler: false - enable_attention_dp: false - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 64 - max_num_tokens: 512 - max_seq_len: 9344 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 20 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [256] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-4p1d-dp8-b64-eplb0-mtp2-c512.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-4p1d-dp8-b64-eplb0-mtp2-c512.yaml deleted file mode 100644 index 5b8f68984e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-4p1d-dp8-b64-eplb0-mtp2-c512.yaml +++ /dev/null @@ -1,137 +0,0 @@ -schema: 2 -name: ctx4_gen1_dp8_batch64_eplb0_mtp2_512 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 4 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: false - disable_overlap_scheduler: true - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - max_batch_size: 8 - max_num_tokens: 8320 - max_seq_len: 8320 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 2 - tensor_parallel_size: 4 - - - decode: - nodes: 1 - workers: 1 - gpus: 8 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: true - max_batch_size: 64 - disable_overlap_scheduler: false - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 64 - max_num_tokens: 650 - max_seq_len: 9344 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 2 - stream_interval: 20 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [512] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-5p1d-dp8-b128-eplb0-mtp0-c1075.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-5p1d-dp8-b128-eplb0-mtp0-c1075.yaml deleted file mode 100644 index 9c8a8e8b10..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-5p1d-dp8-b128-eplb0-mtp0-c1075.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: ctx5_gen1_dp8_batch128_eplb0_mtp0_1075 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 3 - workers: 5 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: false - disable_overlap_scheduler: true - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - max_batch_size: 8 - max_num_tokens: 8320 - max_seq_len: 8320 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 1 - workers: 1 - gpus: 8 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: true - max_batch_size: 128 - disable_overlap_scheduler: false - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 128 - max_num_tokens: 512 - max_seq_len: 9344 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 20 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [1075] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-7p1d-dep8-b384-eplb0-mtp0-c3072.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-7p1d-dep8-b384-eplb0-mtp0-c3072.yaml deleted file mode 100644 index 2687b56fbb..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-7p1d-dep8-b384-eplb0-mtp0-c3072.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: ctx7_gen1_dep8_batch384_eplb0_mtp0_3072 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "b300" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 4 - workers: 7 - gpus: 4 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: false - disable_overlap_scheduler: true - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - max_batch_size: 8 - max_num_tokens: 8320 - max_seq_len: 8320 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 1 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 1 - workers: 1 - gpus: 8 - - env: - NCCL_GRAPH_MIXING_SUPPORT: "0" - OMPI_MCA_coll_ucc_enable: "0" - TLLM_ALL_RANK_LOG: "1" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - UCX_MAX_RMA_RAILS: "1" - UCX_MAX_RNDV_RAILS: "1" - UCX_RNDV_SCHEME: "put_zcopy" - OMPI_MCA_btl: "tcp,self" - OMPI_MCA_pml: "ob1" - TRTLLM_UCX_INTERFACE: "mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1" - - args: - allreduce_strategy: AUTO - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8320 - cuda_graph_config: - enable_padding: true - max_batch_size: 384 - disable_overlap_scheduler: false - enable_attention_dp: true - enable_iter_perf_stats: false - enable_iter_req_stats: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 384 - max_num_tokens: 512 - max_seq_len: 9344 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 20 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: [3072] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - install: false diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml new file mode 100644 index 0000000000..476d3eead9 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml @@ -0,0 +1,461 @@ +# srt-slurm recipes for dsr1/trtllm/b300-fp8/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_. + +schema: 2 + +base: + model: + path: dsr1-fp8 + container: dynamo-trtllm + resources: + gpu_type: b300 + gpus_per_node: 8 + engine: trtllm + roles: + prefill: + gpus: 4 + env: + NCCL_GRAPH_MIXING_SUPPORT: '0' + OMPI_MCA_coll_ucc_enable: '0' + TLLM_ALL_RANK_LOG: '1' + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + UCX_CUDA_IPC_ENABLE_MNNVL: n + UCX_MAX_RMA_RAILS: '1' + UCX_MAX_RNDV_RAILS: '1' + UCX_RNDV_SCHEME: put_zcopy + OMPI_MCA_btl: tcp,self + OMPI_MCA_pml: ob1 + TRTLLM_UCX_INTERFACE: mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1 + args: + allreduce_strategy: AUTO + cache_transceiver_config: + backend: UCX + max_tokens_in_buffer: 8320 + cuda_graph_config: + enable_padding: false + disable_overlap_scheduler: true + enable_attention_dp: true + enable_iter_perf_stats: false + enable_iter_req_stats: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.6 + max_batch_size: 8 + max_num_tokens: 8320 + max_seq_len: 8320 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 1 + pipeline_parallel_size: 1 + print_iter_log: true + tensor_parallel_size: 4 + decode: + gpus: 8 + env: + NCCL_GRAPH_MIXING_SUPPORT: '0' + OMPI_MCA_coll_ucc_enable: '0' + TLLM_ALL_RANK_LOG: '1' + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + UCX_CUDA_IPC_ENABLE_MNNVL: n + UCX_MAX_RMA_RAILS: '1' + UCX_MAX_RNDV_RAILS: '1' + UCX_RNDV_SCHEME: put_zcopy + OMPI_MCA_btl: tcp,self + OMPI_MCA_pml: ob1 + TRTLLM_UCX_INTERFACE: mlx5_0:1,mlx5_1:1,mlx5_10:1,mlx5_11:1,mlx5_16:1,mlx5_17:1,mlx5_20:1,mlx5_21:1,mlx5_22:1,mlx5_23:1,mlx5_4:1,mlx5_5:1,mlx5_8:1,mlx5_9:1,mlx5_2:1,mlx5_3:1 + args: + allreduce_strategy: AUTO + cache_transceiver_config: + backend: UCX + max_tokens_in_buffer: 8320 + cuda_graph_config: + enable_padding: true + disable_overlap_scheduler: false + enable_iter_perf_stats: false + enable_iter_req_stats: false + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.8 + max_seq_len: 9344 + moe_config: + backend: TRTLLM + pipeline_parallel_size: 1 + print_iter_log: true + stream_interval: 20 + tensor_parallel_size: 8 + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + frontend: + type: dynamo + enable_multiple_frontends: false + health_check: + max_attempts: 360 + interval_seconds: 10 + dynamo: + # The unversioned container alias does not identify a Dynamo version or hash. + # Use NATS for a recipe with an unverified Dynamo revision. + request_plane: nats + install: false + +override_disagg_1p1d_dp8_b8_eplb0_mtp3_c72: + name: ctx1_gen1_dp8_batch8_eplb0_mtp3_72 + model: + precision: fp4 + roles: + prefill: + nodes: 1 + workers: 1 + env: + TRTLLM_ENABLE_PDL: '1' + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 1 + workers: 1 + args: + cuda_graph_config: + max_batch_size: 8 + enable_attention_dp: true + max_batch_size: 8 + max_num_tokens: 90 + moe_expert_parallel_size: 1 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: [72] + +override_disagg_1p2d_tp8_b16_eplb0_mtp3_c40: + name: ctx1_gen2_tp8_batch16_eplb0_mtp3_40 + model: + precision: fp4 + roles: + prefill: + nodes: 1 + workers: 1 + env: + TRTLLM_ENABLE_PDL: '1' + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 2 + workers: 2 + env: + TRTLLM_ENABLE_PDL: '1' + args: + cuda_graph_config: + max_batch_size: 16 + enable_attention_dp: false + max_batch_size: 16 + max_num_tokens: 80 + moe_expert_parallel_size: 1 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: [40] + +override_disagg_1p4d_tp8_b1_eplb0_mtp3_c8: + name: ctx1_gen4_tp8_batch1_eplb0_mtp3_8 + model: + precision: fp4 + roles: + prefill: + nodes: 1 + workers: 1 + env: + TRTLLM_ENABLE_PDL: '1' + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 4 + workers: 4 + env: + TRTLLM_ENABLE_PDL: '1' + args: + cuda_graph_config: + max_batch_size: 1 + enable_attention_dp: false + max_batch_size: 1 + max_num_tokens: 4 + moe_expert_parallel_size: 1 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: [5] + +override_disagg_1p4d_tp8_b16_eplb0_mtp0_c64: + name: ctx1_gen4_tp8_batch16_eplb0_mtp0_64 + model: + precision: fp8 + roles: + prefill: + nodes: 1 + workers: 1 + env: + TRTLLM_ENABLE_PDL: '1' + decode: + nodes: 4 + workers: 4 + env: + TRTLLM_ENABLE_PDL: '1' + args: + cuda_graph_config: + max_batch_size: 16 + enable_attention_dp: false + max_batch_size: 16 + max_num_tokens: 512 + moe_expert_parallel_size: 1 + benchmark: + concurrencies: [64] + +override_disagg_1p4d_tp8_b4_eplb0_mtp3_c20: + name: ctx1_gen4_tp8_batch4_eplb0_mtp3_20 + model: + precision: fp4 + roles: + prefill: + nodes: 1 + workers: 1 + env: + TRTLLM_ENABLE_PDL: '1' + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 4 + workers: 4 + env: + TRTLLM_ENABLE_PDL: '1' + args: + cuda_graph_config: + max_batch_size: 4 + enable_attention_dp: false + max_batch_size: 4 + max_num_tokens: 20 + moe_expert_parallel_size: 1 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: [20] + +override_disagg_1p8d_tp8_b1_eplb0_mtp0_c16: + name: ctx1_gen8_tp8_batch2_eplb0_mtp0_16 + model: + precision: fp8 + roles: + prefill: + nodes: 1 + workers: 1 + env: + TRTLLM_ENABLE_PDL: '1' + decode: + nodes: 8 + workers: 8 + env: + TRTLLM_ENABLE_PDL: '1' + args: + cuda_graph_config: + max_batch_size: 1 + enable_attention_dp: false + max_batch_size: 1 + max_num_tokens: 1 + moe_expert_parallel_size: 1 + benchmark: + concurrencies: [10] + +override_disagg_2p1d_dp8_b16_eplb0_mtp3_c144: + name: ctx2_gen1_dp8_batch16_eplb0_mtp3_144 + model: + precision: fp4 + roles: + prefill: + nodes: 1 + workers: 2 + env: + TRTLLM_ENABLE_PDL: '1' + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 1 + workers: 1 + args: + cuda_graph_config: + max_batch_size: 16 + enable_attention_dp: true + max_batch_size: 16 + max_num_tokens: 180 + moe_expert_parallel_size: 1 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: [144] + +override_disagg_2p1d_dp8_b32_eplb0_mtp0_c256: + name: ctx2_gen1_dp8_batch32_eplb0_mtp0_256 + model: + precision: fp8 + roles: + prefill: + nodes: 1 + workers: 2 + env: + TRTLLM_ENABLE_PDL: '1' + decode: + nodes: 1 + workers: 1 + args: + cuda_graph_config: + max_batch_size: 32 + enable_attention_dp: true + max_batch_size: 32 + max_num_tokens: 512 + moe_expert_parallel_size: 1 + benchmark: + concurrencies: [256] + +override_disagg_3p1d_dp8_b64_eplb0_mtp0_c512: + name: ctx3_gen1_dp8_batch64_eplb0_mtp0_512 + model: + precision: fp8 + roles: + prefill: + nodes: 2 + workers: 3 + env: + TRTLLM_ENABLE_PDL: '1' + decode: + nodes: 1 + workers: 1 + args: + cuda_graph_config: + max_batch_size: 64 + enable_attention_dp: true + max_batch_size: 64 + max_num_tokens: 512 + moe_expert_parallel_size: 1 + benchmark: + concurrencies: [512] + +override_disagg_3p5d_tp8_b64_eplb0_mtp0_c256: + name: ctx3_gen5_tp8_batch64_eplb0_mtp0_256 + model: + precision: fp8 + roles: + prefill: + nodes: 2 + workers: 3 + env: + TRTLLM_ENABLE_PDL: '1' + decode: + nodes: 5 + workers: 5 + env: + TRTLLM_ENABLE_PDL: '1' + args: + cuda_graph_config: + max_batch_size: 64 + enable_attention_dp: false + max_batch_size: 64 + max_num_tokens: 512 + moe_expert_parallel_size: 1 + benchmark: + concurrencies: [256] + +override_disagg_4p1d_dp8_b64_eplb0_mtp2_c512: + name: ctx4_gen1_dp8_batch64_eplb0_mtp2_512 + model: + precision: fp4 + roles: + prefill: + nodes: 2 + workers: 4 + env: + TRTLLM_ENABLE_PDL: '1' + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 2 + decode: + nodes: 1 + workers: 1 + args: + cuda_graph_config: + max_batch_size: 64 + enable_attention_dp: true + max_batch_size: 64 + max_num_tokens: 650 + moe_expert_parallel_size: 1 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 2 + benchmark: + concurrencies: [512] + +override_disagg_5p1d_dp8_b128_eplb0_mtp0_c1075: + name: ctx5_gen1_dp8_batch128_eplb0_mtp0_1075 + model: + precision: fp8 + roles: + prefill: + nodes: 3 + workers: 5 + decode: + nodes: 1 + workers: 1 + env: + TRTLLM_ENABLE_PDL: '1' + args: + cuda_graph_config: + max_batch_size: 128 + enable_attention_dp: true + max_batch_size: 128 + max_num_tokens: 512 + moe_expert_parallel_size: 1 + benchmark: + concurrencies: [1075] + +override_disagg_7p1d_dep8_b384_eplb0_mtp0_c3072: + name: ctx7_gen1_dep8_batch384_eplb0_mtp0_3072 + model: + precision: fp8 + roles: + prefill: + nodes: 4 + workers: 7 + env: + TRTLLM_ENABLE_PDL: '1' + decode: + nodes: 1 + workers: 1 + args: + cuda_graph_config: + max_batch_size: 384 + enable_attention_dp: true + max_batch_size: 384 + max_num_tokens: 512 + moe_expert_parallel_size: 8 + benchmark: + concurrencies: [3072] diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-10p1d-dep16-b256-eplb256-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-10p1d-dep16-b256-eplb256-mtp0.yaml deleted file mode 100644 index dc011cc889..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-10p1d-dep16-b256-eplb256-mtp0.yaml +++ /dev/null @@ -1,157 +0,0 @@ -schema: 2 -name: "ctx10_gen1_dep16_batch256_eplb256_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 10 - workers: 10 - - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 4 - - workers: 1 - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 256 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - - 136 - - 144 - - 152 - - 160 - - 168 - - 176 - - 184 - - 192 - - 200 - - 208 - - 216 - - 224 - - 232 - - 240 - - 248 - - 256 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - load_balancer: - num_slots: 256 - layer_updates_per_iter: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "4096" - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-11p1d-dep16-b256-eplb256-mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-11p1d-dep16-b256-eplb256-mtp1.yaml deleted file mode 100644 index 4975070891..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-11p1d-dep16-b256-eplb256-mtp1.yaml +++ /dev/null @@ -1,163 +0,0 @@ -schema: 2 -name: "ctx11_gen1_dep16_batch256_eplb256_mtp1" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 11 - workers: 11 - - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - - decode: - nodes: 4 - - workers: 1 - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 256 - max_num_tokens: 512 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - - 136 - - 144 - - 152 - - 160 - - 168 - - 176 - - 184 - - 192 - - 200 - - 208 - - 216 - - 224 - - 232 - - 240 - - 248 - - 256 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - load_balancer: - num_slots: 256 - layer_updates_per_iter: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "4301" - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0.yaml deleted file mode 100644 index 901eec3517..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0.yaml +++ /dev/null @@ -1,123 +0,0 @@ -schema: 2 -name: "ctx1_gen4_tep8_batch1_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 8 - - workers: 4 - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 1 - max_num_tokens: 1 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "5" - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b16-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b16-eplb0-mtp0.yaml deleted file mode 100644 index 7e70d201cf..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b16-eplb0-mtp0.yaml +++ /dev/null @@ -1,132 +0,0 @@ -schema: 2 -name: "ctx1_gen4_tep8_batch16_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 8 - - workers: 4 - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 16 - max_num_tokens: 16 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 3 - - 4 - - 6 - - 8 - - 9 - - 10 - - 11 - - 14 - - 15 - - 16 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "12x44x76" - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b8-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b8-eplb0-mtp3.yaml deleted file mode 100644 index 4b0a1d57f3..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b8-eplb0-mtp3.yaml +++ /dev/null @@ -1,134 +0,0 @@ -schema: 2 -name: "ctx1_gen4_tep8_batch8_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 8 - - workers: 4 - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 8 - max_num_tokens: 32 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 3 - - 4 - - 5 - - 6 - - 7 - - 8 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "4x8x12x24x48" - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-2p1d-dep32-b8-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-2p1d-dep32-b8-eplb0-mtp0.yaml deleted file mode 100644 index 73ac196bfd..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-2p1d-dep32-b8-eplb0-mtp0.yaml +++ /dev/null @@ -1,123 +0,0 @@ -schema: 2 -name: "ctx2_gen1_dep32_batch8_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 2 - - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 8 - - workers: 1 - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 32 - moe_expert_parallel_size: 32 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 8 - max_num_tokens: 8 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - moe_config: - backend: WIDEEP - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "333" - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-3p1d-dep32-b4-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-3p1d-dep32-b4-eplb0-mtp3.yaml deleted file mode 100644 index 2236c57409..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-3p1d-dep32-b4-eplb0-mtp3.yaml +++ /dev/null @@ -1,128 +0,0 @@ -schema: 2 -name: "ctx3_gen1_dep32_batch4_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 3 - workers: 3 - - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 8 - - workers: 1 - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 32 - moe_expert_parallel_size: 32 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 4 - max_num_tokens: 16 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "180" - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-7p1d-dep16-b64-eplb256-mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-7p1d-dep16-b64-eplb256-mtp1.yaml deleted file mode 100644 index 6f0ea09e6a..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-7p1d-dep16-b64-eplb256-mtp1.yaml +++ /dev/null @@ -1,139 +0,0 @@ -schema: 2 -name: "ctx7_gen1_dep16_batch64_eplb256_mtp1" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 7 - workers: 7 - - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - - decode: - nodes: 4 - - workers: 1 - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 64 - max_num_tokens: 128 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - load_balancer: - num_slots: 256 - layer_updates_per_iter: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1229" - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-7p1d-dep32-b32-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-7p1d-dep32-b32-eplb0-mtp0.yaml deleted file mode 100644 index e0d91dded0..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-7p1d-dep32-b32-eplb0-mtp0.yaml +++ /dev/null @@ -1,126 +0,0 @@ -schema: 2 -name: "ctx7_gen1_dep32_batch32_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 7 - workers: 7 - - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 8 - - workers: 1 - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 32 - moe_expert_parallel_size: 32 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 32 - max_num_tokens: 32 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1229" - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-8p1d-dep16-b128-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-8p1d-dep16-b128-eplb0-mtp0.yaml deleted file mode 100644 index 0a04a36eea..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-8p1d-dep16-b128-eplb0-mtp0.yaml +++ /dev/null @@ -1,138 +0,0 @@ -schema: 2 -name: "ctx8_gen1_dep16_batch128_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 8 - workers: 8 - - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 4 - - workers: 1 - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 128 - max_num_tokens: 128 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "2253" - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-8p1d-dep32-b16-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-8p1d-dep32-b16-eplb0-mtp3.yaml deleted file mode 100644 index 013950ed79..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-8p1d-dep32-b16-eplb0-mtp3.yaml +++ /dev/null @@ -1,130 +0,0 @@ -schema: 2 -name: "ctx8_gen1_dep32_batch16_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 8 - workers: 8 - - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 8 - - workers: 1 - env: - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 32 - moe_expert_parallel_size: 32 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 16 - max_num_tokens: 64 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "666" - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml new file mode 100644 index 0000000000..f64abd99bd --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml @@ -0,0 +1,420 @@ +# srt-slurm recipes for dsr1/trtllm/gb200-fp4/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_. + +schema: 2 + +base: + model: + path: dsr1 + container: dynamo-trtllm + precision: fp4 + resources: + gpu_type: gb200 + gpus_per_node: 4 + engine: trtllm + roles: + prefill: + env: + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + TRTLLM_ENABLE_PDL: '1' + args: + max_batch_size: 2 + max_num_tokens: 16640 + max_seq_len: 8232 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + moe_config: + backend: TRTLLM + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.6 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 16384 + backend: UCX + decode: + env: + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + TRTLLM_ENABLE_PDL: '1' + args: + pipeline_parallel_size: 1 + max_seq_len: 9256 + cuda_graph_config: + enable_padding: true + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + dtype: fp8 + moe_config: + use_low_precision_moe_combine: true + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + cache_transceiver_config: + max_tokens_in_buffer: 16384 + backend: UCX + stream_interval: 100 + num_postprocess_workers: 4 + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + frontend: + type: dynamo + nginx_container: nginx-sqsh + dynamo: + install: false + # The unversioned container alias does not identify a Dynamo version or hash. + # Use NATS for a recipe with an unverified Dynamo revision. + request_plane: nats + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + +override_disagg_10p1d_dep16_b256_eplb256_mtp0: + name: ctx10_gen1_dep16_batch256_eplb256_mtp0 + roles: + prefill: + nodes: 10 + workers: 10 + decode: + nodes: 4 + workers: 1 + args: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 256 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTEDSL + load_balancer: + num_slots: 256 + layer_updates_per_iter: 1 + benchmark: + concurrencies: '4096' + +override_disagg_11p1d_dep16_b256_eplb256_mtp1: + name: ctx11_gen1_dep16_batch256_eplb256_mtp1 + roles: + prefill: + nodes: 11 + workers: 11 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + decode: + nodes: 4 + workers: 1 + args: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 256 + max_num_tokens: 512 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTEDSL + load_balancer: + num_slots: 256 + layer_updates_per_iter: 1 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + benchmark: + concurrencies: '4301' + +override_disagg_1p4d_tep8_b1_eplb0_mtp0: + name: ctx1_gen4_tep8_batch1_eplb0_mtp0 + roles: + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 8 + workers: 4 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 1 + max_num_tokens: 1 + cuda_graph_config: + batch_sizes: [1] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + allreduce_strategy: MNNVL + benchmark: + concurrencies: '5' + +override_disagg_1p4d_tep8_b16_eplb0_mtp0: + name: ctx1_gen4_tep8_batch16_eplb0_mtp0 + roles: + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 8 + workers: 4 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 16 + max_num_tokens: 16 + cuda_graph_config: + batch_sizes: [1, 2, 3, 4, 6, 8, 9, 10, 11, 14, 15, 16] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + allreduce_strategy: MNNVL + benchmark: + concurrencies: 12x44x76 + +override_disagg_1p4d_tep8_b8_eplb0_mtp3: + name: ctx1_gen4_tep8_batch8_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 8 + workers: 4 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 8 + max_num_tokens: 32 + cuda_graph_config: + batch_sizes: [1, 2, 3, 4, 5, 6, 7, 8] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + benchmark: + concurrencies: 4x8x12x24x48 + +override_disagg_2p1d_dep32_b8_eplb0_mtp0: + name: ctx2_gen1_dep32_batch8_eplb0_mtp0 + roles: + prefill: + nodes: 2 + workers: 2 + decode: + nodes: 8 + workers: 1 + args: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 8 + max_num_tokens: 8 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + kv_cache_config: + free_gpu_memory_fraction: 0.75 + moe_config: + backend: WIDEEP + benchmark: + concurrencies: '333' + +override_disagg_3p1d_dep32_b4_eplb0_mtp3: + name: ctx3_gen1_dep32_batch4_eplb0_mtp3 + roles: + prefill: + nodes: 3 + workers: 3 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 8 + workers: 1 + args: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 4 + max_num_tokens: 16 + cuda_graph_config: + batch_sizes: [1, 2, 4] + kv_cache_config: + free_gpu_memory_fraction: 0.75 + moe_config: + backend: CUTEDSL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '180' + +override_disagg_7p1d_dep16_b64_eplb256_mtp1: + name: ctx7_gen1_dep16_batch64_eplb256_mtp1 + roles: + prefill: + nodes: 7 + workers: 7 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + decode: + nodes: 4 + workers: 1 + args: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 64 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTEDSL + load_balancer: + num_slots: 256 + layer_updates_per_iter: 1 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + benchmark: + concurrencies: '1229' + +override_disagg_7p1d_dep32_b32_eplb0_mtp0: + name: ctx7_gen1_dep32_batch32_eplb0_mtp0 + roles: + prefill: + nodes: 7 + workers: 7 + decode: + nodes: 8 + workers: 1 + args: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 32 + max_num_tokens: 32 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32] + kv_cache_config: + free_gpu_memory_fraction: 0.75 + moe_config: + backend: CUTEDSL + benchmark: + concurrencies: '1229' + +override_disagg_8p1d_dep16_b128_eplb0_mtp0: + name: ctx8_gen1_dep16_batch128_eplb0_mtp0 + roles: + prefill: + nodes: 8 + workers: 8 + decode: + nodes: 4 + workers: 1 + args: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 128 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTEDSL + benchmark: + concurrencies: '2253' + +override_disagg_8p1d_dep32_b16_eplb0_mtp3: + name: ctx8_gen1_dep32_batch16_eplb0_mtp3 + roles: + prefill: + nodes: 8 + workers: 8 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 8 + workers: 1 + args: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 16 + max_num_tokens: 64 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + kv_cache_config: + free_gpu_memory_fraction: 0.75 + moe_config: + backend: CUTEDSL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '666' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b1-eplb0-mtp0-c6.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b1-eplb0-mtp0-c6.yaml deleted file mode 100644 index c25597ce46..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b1-eplb0-mtp0-c6.yaml +++ /dev/null @@ -1,125 +0,0 @@ -schema: 2 -name: ctx1_gen3_tep8_batch1_eplb0_mtp0_6 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 8 - - - decode: - nodes: 6 - workers: 3 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - allreduce_strategy: MNNVL - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 1 - max_num_tokens: 1 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['6'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0-c63.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0-c63.yaml deleted file mode 100644 index 64151a38e8..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0-c63.yaml +++ /dev/null @@ -1,127 +0,0 @@ -schema: 2 -name: ctx1_gen3_tep8_batch16_eplb0_mtp0_63 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 8 - - - decode: - nodes: 6 - workers: 3 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - allreduce_strategy: MNNVL - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 16 - max_num_tokens: 16 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['63'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b2-eplb0-mtp3-c6.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b2-eplb0-mtp3-c6.yaml deleted file mode 100644 index e3f15d2954..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b2-eplb0-mtp3-c6.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: ctx1_gen3_tep8_batch2_eplb0_mtp3_6 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 8 - - - decode: - nodes: 6 - workers: 3 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - allreduce_strategy: MNNVL - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 2 - max_num_tokens: 8 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 100 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['6'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b4-eplb0-mtp0-c18.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b4-eplb0-mtp0-c18.yaml deleted file mode 100644 index db3309c1d7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b4-eplb0-mtp0-c18.yaml +++ /dev/null @@ -1,125 +0,0 @@ -schema: 2 -name: ctx1_gen3_tep8_batch4_eplb0_mtp0_18 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 8 - - - decode: - nodes: 6 - workers: 3 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - allreduce_strategy: MNNVL - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 4 - max_num_tokens: 4 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['18'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b4-eplb0-mtp3-c15.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b4-eplb0-mtp3-c15.yaml deleted file mode 100644 index bd33d17265..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b4-eplb0-mtp3-c15.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: ctx1_gen3_tep8_batch4_eplb0_mtp3_15 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 8 - - - decode: - nodes: 6 - workers: 3 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - allreduce_strategy: MNNVL - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 4 - max_num_tokens: 16 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 100 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['15'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-2p1d-dep32-b2-eplb0-mtp3-c90.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-2p1d-dep32-b2-eplb0-mtp3-c90.yaml deleted file mode 100644 index e78a95ad6d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-2p1d-dep32-b2-eplb0-mtp3-c90.yaml +++ /dev/null @@ -1,130 +0,0 @@ -schema: 2 -name: ctx2_gen1_dep32_batch2_eplb0_mtp3_90 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 4 - workers: 2 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 8 - - - decode: - nodes: 8 - workers: 1 - gpus: 32 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 2 - max_num_tokens: 8 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 32 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 100 - tensor_parallel_size: 32 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['90'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-2p1d-dep32-b8-eplb0-mtp0-c333.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-2p1d-dep32-b8-eplb0-mtp0-c333.yaml deleted file mode 100644 index 49da3dc98c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-2p1d-dep32-b8-eplb0-mtp0-c333.yaml +++ /dev/null @@ -1,125 +0,0 @@ -schema: 2 -name: ctx2_gen1_dep32_batch8_eplb0_mtp0_333 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 4 - workers: 2 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 8 - - - decode: - nodes: 8 - workers: 1 - gpus: 32 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - max_batch_size: 8 - max_num_tokens: 8 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 32 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 32 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['333'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep16-b16-eplb0-mtp3-c333.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep16-b16-eplb0-mtp3-c333.yaml deleted file mode 100644 index 9b8f24aaaa..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep16-b16-eplb0-mtp3-c333.yaml +++ /dev/null @@ -1,132 +0,0 @@ -schema: 2 -name: ctx3_gen1_dep16_batch16_eplb0_mtp3_333 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 6 - workers: 3 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 8 - - - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 16 - max_num_tokens: 64 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 100 - tensor_parallel_size: 16 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['333'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep16-b32-eplb0-mtp0-c615.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep16-b32-eplb0-mtp0-c615.yaml deleted file mode 100644 index 2ff39afc48..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep16-b32-eplb0-mtp0-c615.yaml +++ /dev/null @@ -1,128 +0,0 @@ -schema: 2 -name: ctx3_gen1_dep16_batch32_eplb0_mtp0_615 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 6 - workers: 3 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 8 - - - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 32 - max_num_tokens: 32 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 16 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['615'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp3-c666.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp3-c666.yaml deleted file mode 100644 index e932f2ba40..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp3-c666.yaml +++ /dev/null @@ -1,138 +0,0 @@ -schema: 2 -name: ctx3_gen1_dep8_batch64_eplb0_mtp3_666 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 6 - workers: 3 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 8 - - - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 64 - max_num_tokens: 256 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 100 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['666'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-4p1d-dep32-b16-eplb0-mtp0-c666.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-4p1d-dep32-b16-eplb0-mtp0-c666.yaml deleted file mode 100644 index c4488a073d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-4p1d-dep32-b16-eplb0-mtp0-c666.yaml +++ /dev/null @@ -1,126 +0,0 @@ -schema: 2 -name: ctx4_gen1_dep32_batch16_eplb0_mtp0_666 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 8 - workers: 4 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 8 - - - decode: - nodes: 8 - workers: 1 - gpus: 32 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - max_batch_size: 16 - max_num_tokens: 16 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 32 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 32 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['666'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-4p1d-dep32-b8-eplb0-mtp3-c333.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-4p1d-dep32-b8-eplb0-mtp3-c333.yaml deleted file mode 100644 index 96f0c77bf2..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-4p1d-dep32-b8-eplb0-mtp3-c333.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: ctx4_gen1_dep32_batch8_eplb0_mtp3_333 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 8 - workers: 4 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 8 - - - decode: - nodes: 8 - workers: 1 - gpus: 32 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 8 - max_num_tokens: 32 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 32 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 100 - tensor_parallel_size: 32 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['333'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-5p1d-dep16-b32-eplb0-mtp3-c666.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-5p1d-dep16-b32-eplb0-mtp3-c666.yaml deleted file mode 100644 index fbcf7e9f4d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-5p1d-dep16-b32-eplb0-mtp3-c666.yaml +++ /dev/null @@ -1,134 +0,0 @@ -schema: 2 -name: ctx5_gen1_dep16_batch32_eplb0_mtp3_666 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 10 - workers: 5 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 8 - - - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 100 - tensor_parallel_size: 16 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['666'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-5p1d-dep16-b64-eplb0-mtp0-c1229.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-5p1d-dep16-b64-eplb0-mtp0-c1229.yaml deleted file mode 100644 index 8a605d7745..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-5p1d-dep16-b64-eplb0-mtp0-c1229.yaml +++ /dev/null @@ -1,132 +0,0 @@ -schema: 2 -name: ctx5_gen1_dep16_batch64_eplb0_mtp0_1229 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 10 - workers: 5 - gpus: 8 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.4 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 8 - - - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 64 - max_num_tokens: 64 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 16 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['1229'] - req_rate: "inf" - -frontend: - type: "dynamo" - nginx_container: "nginx-sqsh" - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml new file mode 100644 index 0000000000..9d8ab730b4 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml @@ -0,0 +1,514 @@ +# srt-slurm recipes for dsr1/trtllm/gb200-fp8/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_. + +schema: 2 + +base: + model: + path: dsr1-fp8 + container: dynamo-trtllm + precision: fp8 + resources: + gpu_type: gb200 + gpus_per_node: 4 + engine: trtllm + roles: + prefill: + gpus: 8 + env: + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + TRTLLM_ENABLE_PDL: '1' + ENROOT_ALLOW_DEV: 'yes' + NCCL_GRAPH_MIXING_SUPPORT: '0' + args: + cache_transceiver_config: + backend: UCX + max_tokens_in_buffer: 16384 + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.4 + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 8 + pipeline_parallel_size: 1 + print_iter_log: true + tensor_parallel_size: 8 + decode: + env: + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + TRTLLM_ENABLE_PDL: '1' + ENROOT_ALLOW_DEV: 'yes' + NCCL_GRAPH_MIXING_SUPPORT: '0' + TRTLLM_FORCE_COMM_METHOD: NVLINK_TWO_SIDED + ENABLE_CONFIGURABLE_MOE: '1' + args: + cache_transceiver_config: + backend: UCX + max_tokens_in_buffer: 16384 + cuda_graph_config: + enable_padding: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + max_seq_len: 9256 + moe_config: + use_low_precision_moe_combine: true + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + stream_interval: 100 + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + frontend: + type: dynamo + nginx_container: nginx-sqsh + health_check: + max_attempts: 360 + interval_seconds: 10 + dynamo: + install: false + # The unversioned container alias does not identify a Dynamo version or hash. + # Use NATS for a recipe with an unverified Dynamo revision. + request_plane: nats + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + +override_disagg_1p3d_tep8_b1_eplb0_mtp0_c6: + name: ctx1_gen3_tep8_batch1_eplb0_mtp0_6 + roles: + prefill: + nodes: 2 + workers: 1 + decode: + nodes: 6 + workers: 3 + gpus: 8 + args: + allreduce_strategy: MNNVL + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 1 + max_num_tokens: 1 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + benchmark: + concurrencies: ['6'] + +override_disagg_1p3d_tep8_b16_eplb0_mtp0_c63: + name: ctx1_gen3_tep8_batch16_eplb0_mtp0_63 + roles: + prefill: + nodes: 2 + workers: 1 + decode: + nodes: 6 + workers: 3 + gpus: 8 + args: + allreduce_strategy: MNNVL + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 16 + max_num_tokens: 16 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + benchmark: + concurrencies: ['63'] + +override_disagg_1p3d_tep8_b2_eplb0_mtp3_c6: + name: ctx1_gen3_tep8_batch2_eplb0_mtp3_6 + roles: + prefill: + nodes: 2 + workers: 1 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 6 + workers: 3 + gpus: 8 + args: + allreduce_strategy: MNNVL + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 2 + max_num_tokens: 8 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: ['6'] + +override_disagg_1p3d_tep8_b4_eplb0_mtp0_c18: + name: ctx1_gen3_tep8_batch4_eplb0_mtp0_18 + roles: + prefill: + nodes: 2 + workers: 1 + decode: + nodes: 6 + workers: 3 + gpus: 8 + args: + allreduce_strategy: MNNVL + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 4 + max_num_tokens: 4 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + benchmark: + concurrencies: ['18'] + +override_disagg_1p3d_tep8_b4_eplb0_mtp3_c15: + name: ctx1_gen3_tep8_batch4_eplb0_mtp3_15 + roles: + prefill: + nodes: 2 + workers: 1 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 6 + workers: 3 + gpus: 8 + args: + allreduce_strategy: MNNVL + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 4 + max_num_tokens: 16 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: ['15'] + +override_disagg_2p1d_dep32_b2_eplb0_mtp3_c90: + name: ctx2_gen1_dep32_batch2_eplb0_mtp3_90 + roles: + prefill: + nodes: 4 + workers: 2 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 8 + workers: 1 + gpus: 32 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 2 + max_num_tokens: 8 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 32 + tensor_parallel_size: 32 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: ['90'] + +override_disagg_2p1d_dep32_b8_eplb0_mtp0_c333: + name: ctx2_gen1_dep32_batch8_eplb0_mtp0_333 + roles: + prefill: + nodes: 4 + workers: 2 + decode: + nodes: 8 + workers: 1 + gpus: 32 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_batch_size: 8 + max_num_tokens: 8 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 32 + tensor_parallel_size: 32 + benchmark: + concurrencies: ['333'] + +override_disagg_3p1d_dep16_b16_eplb0_mtp3_c333: + name: ctx3_gen1_dep16_batch16_eplb0_mtp3_333 + roles: + prefill: + nodes: 6 + workers: 3 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 16 + max_num_tokens: 64 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: ['333'] + +override_disagg_3p1d_dep16_b32_eplb0_mtp0_c615: + name: ctx3_gen1_dep16_batch32_eplb0_mtp0_615 + roles: + prefill: + nodes: 6 + workers: 3 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 32 + max_num_tokens: 32 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + benchmark: + concurrencies: ['615'] + +override_disagg_3p1d_dep8_b64_eplb0_mtp3_c666: + name: ctx3_gen1_dep8_batch64_eplb0_mtp3_666 + roles: + prefill: + nodes: 6 + workers: 3 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 2 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 64 + max_num_tokens: 256 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: ['666'] + +override_disagg_4p1d_dep32_b16_eplb0_mtp0_c666: + name: ctx4_gen1_dep32_batch16_eplb0_mtp0_666 + roles: + prefill: + nodes: 8 + workers: 4 + decode: + nodes: 8 + workers: 1 + gpus: 32 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_batch_size: 16 + max_num_tokens: 16 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 32 + tensor_parallel_size: 32 + benchmark: + concurrencies: ['666'] + +override_disagg_4p1d_dep32_b8_eplb0_mtp3_c333: + name: ctx4_gen1_dep32_batch8_eplb0_mtp3_333 + roles: + prefill: + nodes: 8 + workers: 4 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 8 + workers: 1 + gpus: 32 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 8 + max_num_tokens: 32 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 32 + tensor_parallel_size: 32 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: ['333'] + +override_disagg_5p1d_dep16_b32_eplb0_mtp3_c666: + name: ctx5_gen1_dep16_batch32_eplb0_mtp3_666 + roles: + prefill: + nodes: 10 + workers: 5 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 32 + max_num_tokens: 128 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: ['666'] + +override_disagg_5p1d_dep16_b64_eplb0_mtp0_c1229: + name: ctx5_gen1_dep16_batch64_eplb0_mtp0_1229 + roles: + prefill: + nodes: 10 + workers: 5 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 64 + max_num_tokens: 64 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + benchmark: + concurrencies: ['1229'] diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep16-b32-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep16-b32-eplb0-mtp3.yaml deleted file mode 100644 index 5b44c7c05f..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep16-b32-eplb0-mtp3.yaml +++ /dev/null @@ -1,134 +0,0 @@ -schema: 2 -name: "ctx10_gen1_dep16_batch32_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 5 - workers: 10 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 4 - - workers: 1 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "666" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp1.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp1.yaml deleted file mode 100644 index 600abc192c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp1.yaml +++ /dev/null @@ -1,162 +0,0 @@ -schema: 2 -name: "ctx10_gen1_dep8_batch256_eplb0_mtp1" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 5 - workers: 10 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - - decode: - nodes: 2 - - workers: 1 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 256 - max_num_tokens: 512 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - - 136 - - 144 - - 152 - - 160 - - 168 - - 176 - - 184 - - 192 - - 200 - - 208 - - 216 - - 224 - - 232 - - 240 - - 248 - - 256 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "2253" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-11p3d-dep4-b256-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-11p3d-dep4-b256-eplb0-mtp0.yaml deleted file mode 100644 index a010b091cc..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-11p3d-dep4-b256-eplb0-mtp0.yaml +++ /dev/null @@ -1,157 +0,0 @@ -schema: 2 -name: "ctx11_gen3_dep4_batch256_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 6 - workers: 11 - gpus: 2 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 3 - - workers: 3 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 256 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - - 136 - - 144 - - 152 - - 160 - - 168 - - 176 - - 184 - - 192 - - 200 - - 208 - - 216 - - 224 - - 232 - - 240 - - 248 - - 256 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "3228" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-13p1d-dep16-b64-eplb256-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-13p1d-dep16-b64-eplb256-mtp3.yaml deleted file mode 100644 index 2c5ed0f516..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-13p1d-dep16-b64-eplb256-mtp3.yaml +++ /dev/null @@ -1,142 +0,0 @@ -schema: 2 -name: "ctx13_gen1_dep16_batch64_eplb256_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 7 - workers: 13 - gpus: 2 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 4 - - workers: 1 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 64 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - load_balancer: - num_slots: 256 - layer_updates_per_iter: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1127" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-14p1d-dep16-b128-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-14p1d-dep16-b128-eplb0-mtp0.yaml deleted file mode 100644 index 22306cad30..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-14p1d-dep16-b128-eplb0-mtp0.yaml +++ /dev/null @@ -1,140 +0,0 @@ -schema: 2 -name: "ctx14_gen1_dep16_batch128_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 7 - workers: 14 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 4 - - workers: 1 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 128 - max_num_tokens: 128 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "2253" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0.yaml deleted file mode 100644 index dadf730a33..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0.yaml +++ /dev/null @@ -1,128 +0,0 @@ -schema: 2 -name: "ctx1_gen3_tep8_batch16_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 6 - - workers: 3 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 16 - max_num_tokens: 16 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "72" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p3d-tep8-b8-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p3d-tep8-b8-eplb0-mtp3.yaml deleted file mode 100644 index 88093b0878..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p3d-tep8-b8-eplb0-mtp3.yaml +++ /dev/null @@ -1,133 +0,0 @@ -schema: 2 -name: "ctx1_gen3_tep8_batch8_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 6 - - workers: 3 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 8 - max_num_tokens: 32 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "33" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0.yaml deleted file mode 100644 index 08eab29c53..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0.yaml +++ /dev/null @@ -1,124 +0,0 @@ -schema: 2 -name: "ctx1_gen4_tep8_batch1_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 8 - - workers: 4 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 1 - max_num_tokens: 1 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "5" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3.yaml deleted file mode 100644 index 6f6da3ebec..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3.yaml +++ /dev/null @@ -1,130 +0,0 @@ -schema: 2 -name: "ctx1_gen4_tep8_batch1_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 8 - - workers: 4 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 1 - max_num_tokens: 4 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "5" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b2-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b2-eplb0-mtp0.yaml deleted file mode 100644 index 8e9576773e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b2-eplb0-mtp0.yaml +++ /dev/null @@ -1,125 +0,0 @@ -schema: 2 -name: "ctx1_gen4_tep8_batch2_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 8 - - workers: 4 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 2 - max_num_tokens: 2 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "12" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3.yaml deleted file mode 100644 index 0f249c08bc..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3.yaml +++ /dev/null @@ -1,133 +0,0 @@ -schema: 2 -name: "ctx1_gen4_tep8_batch4_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 8 - - workers: 4 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 4 - max_num_tokens: 16 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 3 - - 4 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - allreduce_strategy: MNNVL - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "12x24" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p5d-tep4-b4-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p5d-tep4-b4-eplb0-mtp0.yaml deleted file mode 100644 index d139eeac50..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p5d-tep4-b4-eplb0-mtp0.yaml +++ /dev/null @@ -1,125 +0,0 @@ -schema: 2 -name: "ctx1_gen5_tep4_batch4_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 5 - - workers: 5 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 4 - moe_expert_parallel_size: 4 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 4 - max_num_tokens: 4 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "5x15x30" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-4p1d-dep32-b4-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-4p1d-dep32-b4-eplb0-mtp3.yaml deleted file mode 100644 index cc902e3ba1..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-4p1d-dep32-b4-eplb0-mtp3.yaml +++ /dev/null @@ -1,130 +0,0 @@ -schema: 2 -name: "ctx4_gen1_dep32_batch4_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 8 - - workers: 1 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 32 - moe_expert_parallel_size: 32 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 4 - max_num_tokens: 16 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "180" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep32-b16-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep32-b16-eplb0-mtp0.yaml deleted file mode 100644 index 0ec09d8bc7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep32-b16-eplb0-mtp0.yaml +++ /dev/null @@ -1,127 +0,0 @@ -schema: 2 -name: "ctx7_gen1_dep32_batch16_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 4 - workers: 7 - gpus: 2 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 8 - - workers: 1 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 32 - moe_expert_parallel_size: 32 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 16 - max_num_tokens: 16 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "666" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-8p1d-dep32-b8-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-8p1d-dep32-b8-eplb0-mtp3.yaml deleted file mode 100644 index 93f897ecb5..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-8p1d-dep32-b8-eplb0-mtp3.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: "ctx8_gen1_dep32_batch8_eplb0_mtp3" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 4 - workers: 8 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - - decode: - nodes: 8 - - workers: 1 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 32 - moe_expert_parallel_size: 32 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 8 - max_num_tokens: 32 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "308" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-9p1d-dep16-b64-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-9p1d-dep16-b64-eplb0-mtp0.yaml deleted file mode 100644 index 1eee30498c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-9p1d-dep16-b64-eplb0-mtp0.yaml +++ /dev/null @@ -1,133 +0,0 @@ -schema: 2 -name: "ctx9_gen1_dep16_batch64_eplb0_mtp0" - -model: - path: "dsr1" - container: "dynamo-trtllm" - precision: "fp4" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 5 - workers: 9 - gpus: 2 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - tensor_parallel_size: 2 - moe_expert_parallel_size: 2 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - moe_config: - backend: TRTLLM - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.6 - dtype: fp8 - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - - decode: - nodes: 4 - - workers: 1 - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_ENABLE_PDL: "1" - - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 64 - max_num_tokens: 64 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - dtype: fp8 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - cache_transceiver_config: - max_tokens_in_buffer: 16384 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1229" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml new file mode 100644 index 0000000000..9cb11de8d1 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml @@ -0,0 +1,576 @@ +# srt-slurm recipes for dsr1/trtllm/gb300-fp4/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_. + +schema: 2 + +base: + model: + path: dsr1 + container: dynamo-trtllm + precision: fp4 + resources: + gpu_type: gb300 + gpus_per_node: 4 + engine: trtllm + roles: + prefill: + env: + TLLM_OVERRIDE_LAYER_NUM: '61' + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + TRTLLM_ENABLE_PDL: '1' + args: + max_batch_size: 2 + max_num_tokens: 16640 + max_seq_len: 8232 + tensor_parallel_size: 2 + moe_expert_parallel_size: 2 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + moe_config: + backend: TRTLLM + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.6 + dtype: fp8 + cache_transceiver_config: + max_tokens_in_buffer: 16384 + backend: UCX + decode: + env: + TLLM_OVERRIDE_LAYER_NUM: '61' + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + TRTLLM_ENABLE_PDL: '1' + args: + pipeline_parallel_size: 1 + max_seq_len: 9256 + cuda_graph_config: + enable_padding: true + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + dtype: fp8 + moe_config: + use_low_precision_moe_combine: true + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + cache_transceiver_config: + max_tokens_in_buffer: 16384 + backend: UCX + stream_interval: 100 + num_postprocess_workers: 4 + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + frontend: + type: dynamo + enable_multiple_frontends: false + dynamo: + install: false + # The unversioned container alias does not identify a Dynamo version or hash. + # Use NATS for a recipe with an unverified Dynamo revision. + request_plane: nats + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + +override_disagg_10p1d_dep16_b32_eplb0_mtp3: + name: ctx10_gen1_dep16_batch32_eplb0_mtp3 + roles: + prefill: + nodes: 5 + workers: 10 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 4 + workers: 1 + args: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 32 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTEDSL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '666' + +override_disagg_10p1d_dep8_b256_eplb0_mtp1: + name: ctx10_gen1_dep8_batch256_eplb0_mtp1 + roles: + prefill: + nodes: 5 + workers: 10 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + decode: + nodes: 2 + workers: 1 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 256 + max_num_tokens: 512 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + benchmark: + concurrencies: '2253' + +override_disagg_11p3d_dep4_b256_eplb0_mtp0: + name: ctx11_gen3_dep4_batch256_eplb0_mtp0 + roles: + prefill: + nodes: 6 + workers: 11 + gpus: 2 + decode: + nodes: 3 + workers: 3 + args: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 256 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: CUTLASS + benchmark: + concurrencies: '3228' + +override_disagg_13p1d_dep16_b64_eplb256_mtp3: + name: ctx13_gen1_dep16_batch64_eplb256_mtp3 + roles: + prefill: + nodes: 7 + workers: 13 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + gpus: 2 + decode: + nodes: 4 + workers: 1 + args: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 64 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTEDSL + load_balancer: + num_slots: 256 + layer_updates_per_iter: 1 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '1127' + +override_disagg_14p1d_dep16_b128_eplb0_mtp0: + name: ctx14_gen1_dep16_batch128_eplb0_mtp0 + roles: + prefill: + nodes: 7 + workers: 14 + decode: + nodes: 4 + workers: 1 + args: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 128 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTEDSL + benchmark: + concurrencies: '2253' + +override_disagg_1p3d_tep8_b16_eplb0_mtp0: + name: ctx1_gen3_tep8_batch16_eplb0_mtp0 + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 2 + decode: + nodes: 6 + workers: 3 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 16 + max_num_tokens: 16 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + allreduce_strategy: MNNVL + benchmark: + concurrencies: '72' + +override_disagg_1p3d_tep8_b8_eplb0_mtp3: + name: ctx1_gen3_tep8_batch8_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + gpus: 2 + decode: + nodes: 6 + workers: 3 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 8 + max_num_tokens: 32 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + benchmark: + concurrencies: '33' + +override_disagg_1p4d_tep8_b1_eplb0_mtp0: + name: ctx1_gen4_tep8_batch1_eplb0_mtp0 + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 2 + decode: + nodes: 8 + workers: 4 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 1 + max_num_tokens: 1 + cuda_graph_config: + batch_sizes: [1] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + allreduce_strategy: MNNVL + benchmark: + concurrencies: '5' + +override_disagg_1p4d_tep8_b1_eplb0_mtp3: + name: ctx1_gen4_tep8_batch1_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + gpus: 2 + decode: + nodes: 8 + workers: 4 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 1 + max_num_tokens: 4 + cuda_graph_config: + batch_sizes: [1] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + benchmark: + concurrencies: '5' + +override_disagg_1p4d_tep8_b2_eplb0_mtp0: + name: ctx1_gen4_tep8_batch2_eplb0_mtp0 + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 2 + decode: + nodes: 8 + workers: 4 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 2 + max_num_tokens: 2 + cuda_graph_config: + batch_sizes: [1, 2] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + allreduce_strategy: MNNVL + benchmark: + concurrencies: '12' + +override_disagg_1p4d_tep8_b4_eplb0_mtp3: + name: ctx1_gen4_tep8_batch4_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + gpus: 2 + decode: + nodes: 8 + workers: 4 + args: + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 4 + max_num_tokens: 16 + cuda_graph_config: + batch_sizes: [1, 2, 3, 4] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + allreduce_strategy: MNNVL + benchmark: + concurrencies: 12x24 + +override_disagg_1p5d_tep4_b4_eplb0_mtp0: + name: ctx1_gen5_tep4_batch4_eplb0_mtp0 + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 2 + decode: + nodes: 5 + workers: 5 + args: + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 4 + max_num_tokens: 4 + cuda_graph_config: + batch_sizes: [1, 2, 4] + kv_cache_config: + free_gpu_memory_fraction: 0.9 + moe_config: + backend: TRTLLM + benchmark: + concurrencies: 5x15x30 + +override_disagg_4p1d_dep32_b4_eplb0_mtp3: + name: ctx4_gen1_dep32_batch4_eplb0_mtp3 + roles: + prefill: + nodes: 2 + workers: 4 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 8 + workers: 1 + args: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 4 + max_num_tokens: 16 + cuda_graph_config: + batch_sizes: [1, 2, 4] + kv_cache_config: + free_gpu_memory_fraction: 0.75 + moe_config: + backend: CUTEDSL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '180' + +override_disagg_7p1d_dep32_b16_eplb0_mtp0: + name: ctx7_gen1_dep32_batch16_eplb0_mtp0 + roles: + prefill: + nodes: 4 + workers: 7 + gpus: 2 + decode: + nodes: 8 + workers: 1 + args: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 16 + max_num_tokens: 16 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + kv_cache_config: + free_gpu_memory_fraction: 0.75 + moe_config: + backend: CUTEDSL + benchmark: + concurrencies: '666' + +override_disagg_8p1d_dep32_b8_eplb0_mtp3: + name: ctx8_gen1_dep32_batch8_eplb0_mtp3 + roles: + prefill: + nodes: 4 + workers: 8 + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 8 + workers: 1 + args: + tensor_parallel_size: 32 + moe_expert_parallel_size: 32 + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 8 + max_num_tokens: 32 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + kv_cache_config: + free_gpu_memory_fraction: 0.75 + moe_config: + backend: CUTEDSL + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '308' + +override_disagg_9p1d_dep16_b64_eplb0_mtp0: + name: ctx9_gen1_dep16_batch64_eplb0_mtp0 + roles: + prefill: + nodes: 5 + workers: 9 + gpus: 2 + decode: + nodes: 4 + workers: 1 + args: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 64 + max_num_tokens: 64 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: CUTEDSL + benchmark: + concurrencies: '1229' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-10p1d-dep16-b64-eplb0-mtp1-c1229.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-10p1d-dep16-b64-eplb0-mtp1-c1229.yaml deleted file mode 100644 index f39e6ec2f9..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-10p1d-dep16-b64-eplb0-mtp1-c1229.yaml +++ /dev/null @@ -1,141 +0,0 @@ -schema: 2 -name: ctx10_gen1_dep16_batch64_eplb0_mtp1_1229 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 10 - workers: 10 - gpus: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.1 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - tensor_parallel_size: 4 - - - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 64 - max_num_tokens: 128 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - stream_interval: 100 - tensor_parallel_size: 16 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['1229'] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0-c4.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0-c4.yaml deleted file mode 100644 index 0ab2f557eb..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0-c4.yaml +++ /dev/null @@ -1,128 +0,0 @@ -schema: 2 -name: ctx1_gen4_tep8_batch1_eplb0_mtp0_4 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.1 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 8 - workers: 4 - gpus: 8 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - allreduce_strategy: MNNVL - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 1 - max_num_tokens: 1 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['4'] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3-c8.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3-c8.yaml deleted file mode 100644 index 55c20dd5de..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3-c8.yaml +++ /dev/null @@ -1,134 +0,0 @@ -schema: 2 -name: ctx1_gen4_tep8_batch1_eplb0_mtp3_8 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.1 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 4 - - - decode: - nodes: 8 - workers: 4 - gpus: 8 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - allreduce_strategy: MNNVL - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 1 - max_num_tokens: 4 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 100 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['8'] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp0-c24.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp0-c24.yaml deleted file mode 100644 index cee5ba0fcc..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp0-c24.yaml +++ /dev/null @@ -1,128 +0,0 @@ -schema: 2 -name: ctx1_gen4_tep8_batch4_eplb0_mtp0_24 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.1 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 8 - workers: 4 - gpus: 8 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - allreduce_strategy: MNNVL - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 4 - max_num_tokens: 4 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['24'] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3-c24.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3-c24.yaml deleted file mode 100644 index 9b6dfb6235..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3-c24.yaml +++ /dev/null @@ -1,134 +0,0 @@ -schema: 2 -name: ctx1_gen4_tep8_batch4_eplb0_mtp3_24 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.1 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 4 - - - decode: - nodes: 8 - workers: 4 - gpus: 8 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - allreduce_strategy: MNNVL - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 4 - max_num_tokens: 16 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 100 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['24'] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b8-eplb0-mtp0-c36.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b8-eplb0-mtp0-c36.yaml deleted file mode 100644 index bf912e02ea..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b8-eplb0-mtp0-c36.yaml +++ /dev/null @@ -1,129 +0,0 @@ -schema: 2 -name: ctx1_gen4_tep8_batch8_eplb0_mtp0_36 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.1 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 8 - workers: 4 - gpus: 8 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - allreduce_strategy: MNNVL - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 8 - max_num_tokens: 8 - max_seq_len: 9256 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['36'] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-4p1d-dep16-b32-eplb0-mtp0-c666.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-4p1d-dep16-b32-eplb0-mtp0-c666.yaml deleted file mode 100644 index 803171e831..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-4p1d-dep16-b32-eplb0-mtp0-c666.yaml +++ /dev/null @@ -1,131 +0,0 @@ -schema: 2 -name: ctx4_gen1_dep16_batch32_eplb0_mtp0_666 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 4 - workers: 4 - gpus: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.1 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 32 - max_num_tokens: 32 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 16 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['666'] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-6p1d-dep32-b16-eplb0-mtp0-c512.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-6p1d-dep32-b16-eplb0-mtp0-c512.yaml deleted file mode 100644 index 2736b28f56..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-6p1d-dep32-b16-eplb0-mtp0-c512.yaml +++ /dev/null @@ -1,129 +0,0 @@ -schema: 2 -name: ctx6_gen1_dep32_batch16_eplb0_mtp0_512 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 6 - workers: 6 - gpus: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.1 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 8 - workers: 1 - gpus: 32 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - max_batch_size: 16 - max_num_tokens: 16 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 32 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 32 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['512'] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-6p1d-dep32-b8-eplb0-mtp3-c333.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-6p1d-dep32-b8-eplb0-mtp3-c333.yaml deleted file mode 100644 index 52974739c3..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-6p1d-dep32-b8-eplb0-mtp3-c333.yaml +++ /dev/null @@ -1,134 +0,0 @@ -schema: 2 -name: ctx6_gen1_dep32_batch8_eplb0_mtp3_333 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 6 - workers: 6 - gpus: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.1 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 4 - - - decode: - nodes: 8 - workers: 1 - gpus: 32 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 8 - max_num_tokens: 32 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 32 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 100 - tensor_parallel_size: 32 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['333'] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep16-b64-eplb0-mtp0-c1229.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep16-b64-eplb0-mtp0-c1229.yaml deleted file mode 100644 index 5acd5089d1..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep16-b64-eplb0-mtp0-c1229.yaml +++ /dev/null @@ -1,135 +0,0 @@ -schema: 2 -name: ctx7_gen1_dep16_batch64_eplb0_mtp0_1229 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 7 - workers: 7 - gpus: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.1 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 64 - max_num_tokens: 64 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 16 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['1229'] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep8-b128-eplb0-mtp1-c1229.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep8-b128-eplb0-mtp1-c1229.yaml deleted file mode 100644 index 4c449203b0..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep8-b128-eplb0-mtp1-c1229.yaml +++ /dev/null @@ -1,149 +0,0 @@ -schema: 2 -name: ctx7_gen1_dep8_batch128_eplb0_mtp1_1229 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 7 - workers: 7 - gpus: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8192 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.1 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - tensor_parallel_size: 4 - - - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8192 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 128 - max_num_tokens: 256 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - stream_interval: 100 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['1229'] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep8-b256-eplb0-mtp0-c2151.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep8-b256-eplb0-mtp0-c2151.yaml deleted file mode 100644 index 587ba16789..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep8-b256-eplb0-mtp0-c2151.yaml +++ /dev/null @@ -1,159 +0,0 @@ -schema: 2 -name: ctx7_gen1_dep8_batch256_eplb0_mtp0_2151 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 7 - workers: 7 - gpus: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.1 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 4 - - - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - - 136 - - 144 - - 152 - - 160 - - 168 - - 176 - - 184 - - 192 - - 200 - - 208 - - 216 - - 224 - - 232 - - 240 - - 248 - - 256 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 256 - max_num_tokens: 256 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['2151'] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-8p1d-dep16-b32-eplb0-mtp3-c666.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-8p1d-dep16-b32-eplb0-mtp3-c666.yaml deleted file mode 100644 index a1d7b254c3..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-8p1d-dep16-b32-eplb0-mtp3-c666.yaml +++ /dev/null @@ -1,137 +0,0 @@ -schema: 2 -name: ctx8_gen1_dep16_batch32_eplb0_mtp3_666 - -model: - path: "dsr1-fp8" - container: "dynamo-trtllm" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - -engine: trtllm -roles: - prefill: - nodes: 8 - workers: 8 - gpus: 4 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.1 - max_batch_size: 2 - max_num_tokens: 16384 - max_seq_len: 8232 - moe_config: - backend: DEEPGEMM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - tensor_parallel_size: 4 - - - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - TLLM_OVERRIDE_LAYER_NUM: "61" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - ENROOT_ALLOW_DEV: "yes" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TRTLLM_FORCE_COMM_METHOD: "NVLINK_TWO_SIDED" - ENABLE_CONFIGURABLE_MOE: "1" - - args: - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 9256 - moe_config: - backend: DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - stream_interval: 100 - tensor_parallel_size: 16 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: ['666'] - req_rate: "inf" - -frontend: - type: "dynamo" - - enable_multiple_frontends: false - - -health_check: - max_attempts: 360 - interval_seconds: 10 - -dynamo: - install: false - # The unversioned container alias does not identify a Dynamo version or hash. - # Use NATS for a recipe with an unverified Dynamo revision. - request_plane: "nats" -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml new file mode 100644 index 0000000000..49b34c9a58 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml @@ -0,0 +1,540 @@ +# srt-slurm recipes for dsr1/trtllm/gb300-fp8/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_. + +schema: 2 + +base: + model: + path: dsr1-fp8 + container: dynamo-trtllm + precision: fp8 + resources: + gpu_type: gb300 + gpus_per_node: 4 + engine: trtllm + roles: + prefill: + gpus: 4 + env: + TLLM_OVERRIDE_LAYER_NUM: '61' + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + TRTLLM_ENABLE_PDL: '1' + ENROOT_ALLOW_DEV: 'yes' + NCCL_GRAPH_MIXING_SUPPORT: '0' + args: + cache_transceiver_config: + backend: UCX + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + free_gpu_memory_fraction: 0.1 + max_batch_size: 2 + max_num_tokens: 16384 + max_seq_len: 8232 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 4 + pipeline_parallel_size: 1 + print_iter_log: true + tensor_parallel_size: 4 + decode: + env: + TLLM_OVERRIDE_LAYER_NUM: '61' + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + TRTLLM_ENABLE_PDL: '1' + ENROOT_ALLOW_DEV: 'yes' + NCCL_GRAPH_MIXING_SUPPORT: '0' + TRTLLM_FORCE_COMM_METHOD: NVLINK_TWO_SIDED + ENABLE_CONFIGURABLE_MOE: '1' + args: + cache_transceiver_config: + backend: UCX + cuda_graph_config: + enable_padding: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + max_seq_len: 9256 + moe_config: + use_low_precision_moe_combine: true + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + stream_interval: 100 + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + frontend: + type: dynamo + enable_multiple_frontends: false + health_check: + max_attempts: 360 + interval_seconds: 10 + dynamo: + install: false + # The unversioned container alias does not identify a Dynamo version or hash. + # Use NATS for a recipe with an unverified Dynamo revision. + request_plane: nats + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + +override_disagg_10p1d_dep16_b64_eplb0_mtp1_c1229: + name: ctx10_gen1_dep16_batch64_eplb0_mtp1_1229 + roles: + prefill: + nodes: 10 + workers: 10 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 64 + max_num_tokens: 128 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 16 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + tensor_parallel_size: 16 + benchmark: + concurrencies: ['1229'] + +override_disagg_1p4d_tep8_b1_eplb0_mtp0_c4: + name: ctx1_gen4_tep8_batch1_eplb0_mtp0_4 + roles: + prefill: + nodes: 1 + workers: 1 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + decode: + nodes: 8 + workers: 4 + gpus: 8 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 1 + max_num_tokens: 1 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + allreduce_strategy: MNNVL + benchmark: + concurrencies: ['4'] + +override_disagg_1p4d_tep8_b1_eplb0_mtp3_c8: + name: ctx1_gen4_tep8_batch1_eplb0_mtp3_8 + roles: + prefill: + nodes: 1 + workers: 1 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 8 + workers: 4 + gpus: 8 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 1 + max_num_tokens: 4 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + tensor_parallel_size: 8 + allreduce_strategy: MNNVL + benchmark: + concurrencies: ['8'] + +override_disagg_1p4d_tep8_b4_eplb0_mtp0_c24: + name: ctx1_gen4_tep8_batch4_eplb0_mtp0_24 + roles: + prefill: + nodes: 1 + workers: 1 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + decode: + nodes: 8 + workers: 4 + gpus: 8 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 4 + max_num_tokens: 4 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + allreduce_strategy: MNNVL + benchmark: + concurrencies: ['24'] + +override_disagg_1p4d_tep8_b4_eplb0_mtp3_c24: + name: ctx1_gen4_tep8_batch4_eplb0_mtp3_24 + roles: + prefill: + nodes: 1 + workers: 1 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 8 + workers: 4 + gpus: 8 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 4 + max_num_tokens: 16 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + tensor_parallel_size: 8 + allreduce_strategy: MNNVL + benchmark: + concurrencies: ['24'] + +override_disagg_1p4d_tep8_b8_eplb0_mtp0_c36: + name: ctx1_gen4_tep8_batch8_eplb0_mtp0_36 + roles: + prefill: + nodes: 1 + workers: 1 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + decode: + nodes: 8 + workers: 4 + gpus: 8 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 8 + max_num_tokens: 8 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + allreduce_strategy: MNNVL + benchmark: + concurrencies: ['36'] + +override_disagg_4p1d_dep16_b32_eplb0_mtp0_c666: + name: ctx4_gen1_dep16_batch32_eplb0_mtp0_666 + roles: + prefill: + nodes: 4 + workers: 4 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 32 + max_num_tokens: 32 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + benchmark: + concurrencies: ['666'] + +override_disagg_6p1d_dep32_b16_eplb0_mtp0_c512: + name: ctx6_gen1_dep32_batch16_eplb0_mtp0_512 + roles: + prefill: + nodes: 6 + workers: 6 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + decode: + nodes: 8 + workers: 1 + gpus: 32 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_batch_size: 16 + max_num_tokens: 16 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 32 + tensor_parallel_size: 32 + benchmark: + concurrencies: ['512'] + +override_disagg_6p1d_dep32_b8_eplb0_mtp3_c333: + name: ctx6_gen1_dep32_batch8_eplb0_mtp3_333 + roles: + prefill: + nodes: 6 + workers: 6 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 8 + workers: 1 + gpus: 32 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 8 + max_num_tokens: 32 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 32 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + tensor_parallel_size: 32 + benchmark: + concurrencies: ['333'] + +override_disagg_7p1d_dep16_b64_eplb0_mtp0_c1229: + name: ctx7_gen1_dep16_batch64_eplb0_mtp0_1229 + roles: + prefill: + nodes: 7 + workers: 7 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 64 + max_num_tokens: 64 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + benchmark: + concurrencies: ['1229'] + +override_disagg_7p1d_dep8_b128_eplb0_mtp1_c1229: + name: ctx7_gen1_dep8_batch128_eplb0_mtp1_1229 + roles: + prefill: + nodes: 7 + workers: 7 + args: + cache_transceiver_config: + max_tokens_in_buffer: 8192 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + decode: + nodes: 2 + workers: 1 + gpus: 8 + args: + cache_transceiver_config: + max_tokens_in_buffer: 8192 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 128 + max_num_tokens: 256 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 8 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + tensor_parallel_size: 8 + benchmark: + concurrencies: ['1229'] + +override_disagg_7p1d_dep8_b256_eplb0_mtp0_c2151: + name: ctx7_gen1_dep8_batch256_eplb0_mtp0_2151 + roles: + prefill: + nodes: 7 + workers: 7 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + decode: + nodes: 2 + workers: 1 + gpus: 8 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 256 + max_num_tokens: 256 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + benchmark: + concurrencies: ['2151'] + +override_disagg_8p1d_dep16_b32_eplb0_mtp3_c666: + name: ctx8_gen1_dep16_batch32_eplb0_mtp3_666 + roles: + prefill: + nodes: 8 + workers: 8 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cache_transceiver_config: + max_tokens_in_buffer: 16384 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 32 + max_num_tokens: 128 + moe_config: + backend: DEEPGEMM + moe_expert_parallel_size: 16 + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + tensor_parallel_size: 16 + benchmark: + concurrencies: ['666'] diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p1d-dep16-b4-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p1d-dep16-b4-eplb0-mtp3.yaml deleted file mode 100644 index 8b8846c8b0..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p1d-dep16-b4-eplb0-mtp3.yaml +++ /dev/null @@ -1,106 +0,0 @@ -schema: 2 -name: h100_8k1k_ctx1dep16_gen1dep16_batch4_eplb0_mtp3 -model: - path: DeepSeek-R1-0528 - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post3" - precision: fp8 -resources: - gpu_type: h100 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - env: - UCX_CUDA_IPC_ENABLE_MNNVL: n - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' - TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP - args: - max_batch_size: 1 - max_num_tokens: 8224 - max_seq_len: 8232 - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - enable_chunked_prefill: true - moe_config: - backend: WIDEEP - max_num_tokens: 16384 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.3 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8256 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - decode: - nodes: 2 - workers: 1 - env: - NCCL_NVLS_ENABLE: '0' - UCX_CUDA_IPC_ENABLE_MNNVL: n - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' - TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 4 - max_num_tokens: 128 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: WIDEEP - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8256 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: '77' - req_rate: inf -frontend: - type: dynamo - enable_multiple_frontends: false -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p2d-tep16-b32-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p2d-tep16-b32-eplb0-mtp3.yaml deleted file mode 100644 index 51cdf9c090..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p2d-tep16-b32-eplb0-mtp3.yaml +++ /dev/null @@ -1,108 +0,0 @@ -schema: 2 -name: h100_8k1k_ctx1dep16_gen2tep16_batch32_eplb0_mtp3 -model: - path: DeepSeek-R1-0528 - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post3" - precision: fp8 -resources: - gpu_type: h100 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - env: - UCX_CUDA_IPC_ENABLE_MNNVL: n - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' - TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP - args: - max_batch_size: 1 - max_num_tokens: 8224 - max_seq_len: 8232 - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - enable_chunked_prefill: false - moe_config: - backend: WIDEEP - max_num_tokens: 16384 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.3 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8256 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - decode: - nodes: 4 - workers: 2 - env: - NCCL_NVLS_ENABLE: '0' - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - UCX_CUDA_IPC_ENABLE_MNNVL: n - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 32 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8256 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: '78' - req_rate: inf -frontend: - type: dynamo - enable_multiple_frontends: false -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p2d-tep16-b64-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p2d-tep16-b64-eplb0-mtp0.yaml deleted file mode 100644 index 8a06e91682..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p2d-tep16-b64-eplb0-mtp0.yaml +++ /dev/null @@ -1,107 +0,0 @@ - -schema: 2 -name: "h100_8k1k_ctx1dep16_gen2tep16_batch64_eplb0_mtp0" - -model: - path: "DeepSeek-R1-0528" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post3" - precision: "fp8" - -resources: - gpu_type: "h100" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - env: - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: "1" - TRTLLM_FORCE_ALLTOALL_METHOD: "DeepEP" - - args: - max_batch_size: 1 - max_num_tokens: 8224 - max_seq_len: 8232 - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - enable_chunked_prefill: false - moe_config: - backend: WIDEEP - max_num_tokens: 16384 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.3 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8256 - - - decode: - nodes: 4 - workers: 2 - env: - NCCL_NVLS_ENABLE: "0" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TLLM_LOG_LEVEL: "INFO" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 64 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64] - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8256 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "154" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # There are errors about colliding on port 8080, and others. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b1-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b1-eplb0-mtp0.yaml deleted file mode 100644 index 7e27a3b932..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b1-eplb0-mtp0.yaml +++ /dev/null @@ -1,99 +0,0 @@ -schema: 2 -name: h100_8k1k_ctx1dep16_gen3tep16_batch1_eplb0_mtp0 -model: - path: DeepSeek-R1-0528 - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post3" - precision: fp8 -resources: - gpu_type: h100 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - env: - UCX_CUDA_IPC_ENABLE_MNNVL: n - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' - TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP - args: - max_batch_size: 1 - max_num_tokens: 8224 - max_seq_len: 8232 - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - enable_chunked_prefill: false - moe_config: - backend: WIDEEP - max_num_tokens: 16384 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.3 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8256 - decode: - nodes: 6 - workers: 3 - env: - NCCL_NVLS_ENABLE: '0' - UCX_CUDA_IPC_ENABLE_MNNVL: n - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 1 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8256 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: '6' - req_rate: inf -frontend: - type: dynamo - enable_multiple_frontends: false -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b1-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b1-eplb0-mtp3.yaml deleted file mode 100644 index 1f3fd709d9..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b1-eplb0-mtp3.yaml +++ /dev/null @@ -1,104 +0,0 @@ -schema: 2 -name: h100_8k1k_ctx1dep16_gen3tep16_batch1_eplb0_mtp3 -model: - path: DeepSeek-R1-0528 - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post3" - precision: fp8 -resources: - gpu_type: h100 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - env: - UCX_CUDA_IPC_ENABLE_MNNVL: n - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' - TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP - args: - max_batch_size: 1 - max_num_tokens: 8224 - max_seq_len: 8232 - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - enable_chunked_prefill: true - moe_config: - backend: WIDEEP - max_num_tokens: 16384 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.3 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8256 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - decode: - nodes: 6 - workers: 3 - env: - NCCL_NVLS_ENABLE: '0' - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - UCX_CUDA_IPC_ENABLE_MNNVL: n - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 1 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8256 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: '6' - req_rate: inf -frontend: - type: dynamo - enable_multiple_frontends: false -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b2-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b2-eplb0-mtp0.yaml deleted file mode 100644 index 740de40de0..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b2-eplb0-mtp0.yaml +++ /dev/null @@ -1,107 +0,0 @@ - -schema: 2 -name: "h100_8k1k_ctx1dep16_gen3tep16_batch2_eplb0_mtp0" - -model: - path: "DeepSeek-R1-0528" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post3" - precision: "fp8" - -resources: - gpu_type: "h100" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - env: - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: "1" - TRTLLM_FORCE_ALLTOALL_METHOD: "DeepEP" - - args: - max_batch_size: 1 - max_num_tokens: 8224 - max_seq_len: 8232 - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - enable_chunked_prefill: false - moe_config: - backend: WIDEEP - max_num_tokens: 16384 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.3 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8256 - - - decode: - nodes: 6 - workers: 3 - env: - NCCL_NVLS_ENABLE: "0" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TLLM_LOG_LEVEL: "INFO" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 2 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4] - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8256 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "9" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # There are errors about colliding on port 8080, and others. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b2-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b2-eplb0-mtp3.yaml deleted file mode 100644 index 03ac6a374b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b2-eplb0-mtp3.yaml +++ /dev/null @@ -1,104 +0,0 @@ -schema: 2 -name: h100_8k1k_ctx1dep16_gen3tep16_batch2_eplb0_mtp3 -model: - path: DeepSeek-R1-0528 - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post3" - precision: fp8 -resources: - gpu_type: h100 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - env: - UCX_CUDA_IPC_ENABLE_MNNVL: n - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' - TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP - args: - max_batch_size: 1 - max_num_tokens: 8224 - max_seq_len: 8232 - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - enable_chunked_prefill: true - moe_config: - backend: WIDEEP - max_num_tokens: 16384 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.3 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8256 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - decode: - nodes: 6 - workers: 3 - env: - NCCL_NVLS_ENABLE: '0' - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - UCX_CUDA_IPC_ENABLE_MNNVL: n - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 2 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8256 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: '9' - req_rate: inf -frontend: - type: dynamo - enable_multiple_frontends: false -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b8-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b8-eplb0-mtp0.yaml deleted file mode 100644 index b22aa1922c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b8-eplb0-mtp0.yaml +++ /dev/null @@ -1,107 +0,0 @@ - -schema: 2 -name: "h100_8k1k_ctx1dep16_gen3tep16_batch8_eplb0_mtp0" - -model: - path: "DeepSeek-R1-0528" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post3" - precision: "fp8" - -resources: - gpu_type: "h100" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - env: - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TLLM_LOG_LEVEL: "INFO" - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: "1" - TRTLLM_FORCE_ALLTOALL_METHOD: "DeepEP" - - args: - max_batch_size: 1 - max_num_tokens: 8224 - max_seq_len: 8232 - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - enable_chunked_prefill: true - moe_config: - backend: WIDEEP - max_num_tokens: 16384 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.3 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8256 - - - decode: - nodes: 6 - workers: 3 - env: - NCCL_NVLS_ENABLE: "0" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - TLLM_LOG_LEVEL: "INFO" - UCX_CUDA_IPC_ENABLE_MNNVL: "n" - - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 8 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8] - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8256 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "30" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # There are errors about colliding on port 8080, and others. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b8-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b8-eplb0-mtp3.yaml deleted file mode 100644 index 4e1eb0d921..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b8-eplb0-mtp3.yaml +++ /dev/null @@ -1,105 +0,0 @@ -schema: 2 -name: h100_8k1k_ctx1dep16_gen3tep16_batch8_eplb0_mtp3 -model: - path: DeepSeek-R1-0528 - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post3" - precision: fp8 -resources: - gpu_type: h100 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 1 - env: - UCX_CUDA_IPC_ENABLE_MNNVL: n - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' - TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP - args: - max_batch_size: 1 - max_num_tokens: 8224 - max_seq_len: 8232 - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - enable_chunked_prefill: false - moe_config: - backend: WIDEEP - max_num_tokens: 16384 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.3 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8256 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - decode: - nodes: 6 - workers: 3 - env: - NCCL_NVLS_ENABLE: '0' - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - UCX_CUDA_IPC_ENABLE_MNNVL: n - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 8 - max_num_tokens: 256 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8256 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: '30' - req_rate: inf -frontend: - type: dynamo - enable_multiple_frontends: false -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-2p1d-dep16-b16-eplb0-mtp0.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-2p1d-dep16-b16-eplb0-mtp0.yaml deleted file mode 100644 index 91dce0d59c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-2p1d-dep16-b16-eplb0-mtp0.yaml +++ /dev/null @@ -1,102 +0,0 @@ -schema: 2 -name: h100_8k1k_ctx2dep16_gen1dep16_batch16_eplb0_mtp0 -model: - path: DeepSeek-R1-0528 - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post3" - precision: fp8 -resources: - gpu_type: h100 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 4 - workers: 2 - env: - UCX_CUDA_IPC_ENABLE_MNNVL: n - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' - TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP - args: - max_batch_size: 1 - max_num_tokens: 8224 - max_seq_len: 8232 - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - enable_chunked_prefill: false - moe_config: - backend: WIDEEP - max_num_tokens: 16384 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.3 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8256 - decode: - nodes: 2 - workers: 1 - env: - NCCL_NVLS_ENABLE: '0' - UCX_CUDA_IPC_ENABLE_MNNVL: n - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' - TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - pipeline_parallel_size: 1 - max_batch_size: 16 - max_num_tokens: 128 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: WIDEEP - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8256 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: '308' - req_rate: inf -frontend: - type: dynamo - enable_multiple_frontends: false -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-2p1d-dep16-b8-eplb0-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-2p1d-dep16-b8-eplb0-mtp3.yaml deleted file mode 100644 index 5eb9fdc23e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-2p1d-dep16-b8-eplb0-mtp3.yaml +++ /dev/null @@ -1,107 +0,0 @@ -schema: 2 -name: h100_8k1k_ctx2dep16_gen1dep16_batch8_eplb0_mtp3 -model: - path: DeepSeek-R1-0528 - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post3" - precision: fp8 -resources: - gpu_type: h100 - gpus_per_node: 8 -engine: trtllm -roles: - prefill: - nodes: 4 - workers: 2 - env: - UCX_CUDA_IPC_ENABLE_MNNVL: n - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' - TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP - args: - max_batch_size: 1 - max_num_tokens: 8224 - max_seq_len: 8232 - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - pipeline_parallel_size: 1 - print_iter_log: true - cuda_graph_config: - disable_overlap_scheduler: true - enable_chunked_prefill: true - moe_config: - backend: WIDEEP - max_num_tokens: 16384 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.3 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 8256 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - decode: - nodes: 2 - workers: 1 - env: - NCCL_NVLS_ENABLE: '0' - UCX_CUDA_IPC_ENABLE_MNNVL: n - TRTLLM_ENABLE_PDL: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - TLLM_LOG_LEVEL: INFO - TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' - TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP - args: - tensor_parallel_size: 16 - moe_expert_parallel_size: 16 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - pipeline_parallel_size: 1 - max_batch_size: 8 - max_num_tokens: 128 - max_seq_len: 9256 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - print_iter_log: true - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - dtype: fp8 - moe_config: - backend: WIDEEP - use_low_precision_moe_combine: true - cache_transceiver_config: - max_tokens_in_buffer: 8256 - backend: UCX - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: '154' - req_rate: inf -frontend: - type: dynamo - enable_multiple_frontends: false -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml new file mode 100644 index 0000000000..3274bdf09e --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml @@ -0,0 +1,391 @@ +# srt-slurm recipes for dsr1/trtllm/h100-fp8/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_. + +schema: 2 + +base: + model: + path: DeepSeek-R1-0528 + container: nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post3 + precision: fp8 + resources: + gpu_type: h100 + gpus_per_node: 8 + engine: trtllm + roles: + prefill: + env: + UCX_CUDA_IPC_ENABLE_MNNVL: n + TRTLLM_ENABLE_PDL: '1' + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + TLLM_LOG_LEVEL: INFO + TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' + TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP + args: + max_batch_size: 1 + max_num_tokens: 8224 + max_seq_len: 8232 + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + enable_attention_dp: true + pipeline_parallel_size: 1 + print_iter_log: true + cuda_graph_config: null + disable_overlap_scheduler: true + moe_config: + backend: WIDEEP + max_num_tokens: 16384 + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.3 + dtype: fp8 + cache_transceiver_config: + backend: UCX + max_tokens_in_buffer: 8256 + decode: + env: + NCCL_NVLS_ENABLE: '0' + UCX_CUDA_IPC_ENABLE_MNNVL: n + TRTLLM_ENABLE_PDL: '1' + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + TLLM_LOG_LEVEL: INFO + args: + tensor_parallel_size: 16 + moe_expert_parallel_size: 16 + pipeline_parallel_size: 1 + max_seq_len: 9256 + cuda_graph_config: + enable_padding: true + print_iter_log: true + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.9 + dtype: fp8 + moe_config: + use_low_precision_moe_combine: true + cache_transceiver_config: + max_tokens_in_buffer: 8256 + backend: UCX + stream_interval: 100 + num_postprocess_workers: 4 + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + frontend: + type: dynamo + enable_multiple_frontends: false + dynamo: + install: false + # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. + # Use NATS for a recipe prior to Dynamo commit 39d2a68. + request_plane: nats + +override_disagg_1p1d_dep16_b4_eplb0_mtp3: + name: h100_8k1k_ctx1dep16_gen1dep16_batch4_eplb0_mtp3 + roles: + prefill: + nodes: 2 + workers: 1 + args: + enable_chunked_prefill: true + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 2 + workers: 1 + env: + TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' + TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP + args: + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 4 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4] + moe_config: + backend: WIDEEP + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '77' + +override_disagg_1p2d_tep16_b32_eplb0_mtp3: + name: h100_8k1k_ctx1dep16_gen2tep16_batch32_eplb0_mtp3 + roles: + prefill: + nodes: 2 + workers: 1 + args: + enable_chunked_prefill: false + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 4 + workers: 2 + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 32 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32] + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '78' + +# (frontend.enable_multiple_frontends) There are errors about colliding on port 8080, and others. +override_disagg_1p2d_tep16_b64_eplb0_mtp0: + name: h100_8k1k_ctx1dep16_gen2tep16_batch64_eplb0_mtp0 + roles: + prefill: + nodes: 2 + workers: 1 + args: + enable_chunked_prefill: false + decode: + nodes: 4 + workers: 2 + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 64 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64] + moe_config: + backend: CUTLASS + benchmark: + concurrencies: '154' + +override_disagg_1p3d_tep16_b1_eplb0_mtp0: + name: h100_8k1k_ctx1dep16_gen3tep16_batch1_eplb0_mtp0 + roles: + prefill: + nodes: 2 + workers: 1 + args: + enable_chunked_prefill: false + decode: + nodes: 6 + workers: 3 + env: + TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 1 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4] + moe_config: + backend: CUTLASS + benchmark: + concurrencies: '6' + +override_disagg_1p3d_tep16_b1_eplb0_mtp3: + name: h100_8k1k_ctx1dep16_gen3tep16_batch1_eplb0_mtp3 + roles: + prefill: + nodes: 2 + workers: 1 + args: + enable_chunked_prefill: true + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 6 + workers: 3 + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 1 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4] + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '6' + +# (frontend.enable_multiple_frontends) There are errors about colliding on port 8080, and others. +override_disagg_1p3d_tep16_b2_eplb0_mtp0: + name: h100_8k1k_ctx1dep16_gen3tep16_batch2_eplb0_mtp0 + roles: + prefill: + nodes: 2 + workers: 1 + args: + enable_chunked_prefill: false + decode: + nodes: 6 + workers: 3 + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 2 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4] + moe_config: + backend: CUTLASS + benchmark: + concurrencies: '9' + +override_disagg_1p3d_tep16_b2_eplb0_mtp3: + name: h100_8k1k_ctx1dep16_gen3tep16_batch2_eplb0_mtp3 + roles: + prefill: + nodes: 2 + workers: 1 + args: + enable_chunked_prefill: true + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 6 + workers: 3 + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 2 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4] + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '9' + +# (frontend.enable_multiple_frontends) There are errors about colliding on port 8080, and others. +override_disagg_1p3d_tep16_b8_eplb0_mtp0: + name: h100_8k1k_ctx1dep16_gen3tep16_batch8_eplb0_mtp0 + roles: + prefill: + nodes: 2 + workers: 1 + args: + enable_chunked_prefill: true + decode: + nodes: 6 + workers: 3 + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 8 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + moe_config: + backend: CUTLASS + benchmark: + concurrencies: '30' + +override_disagg_1p3d_tep16_b8_eplb0_mtp3: + name: h100_8k1k_ctx1dep16_gen3tep16_batch8_eplb0_mtp3 + roles: + prefill: + nodes: 2 + workers: 1 + args: + enable_chunked_prefill: false + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 6 + workers: 3 + args: + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + max_batch_size: 8 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '30' + +override_disagg_2p1d_dep16_b16_eplb0_mtp0: + name: h100_8k1k_ctx2dep16_gen1dep16_batch16_eplb0_mtp0 + roles: + prefill: + nodes: 4 + workers: 2 + args: + enable_chunked_prefill: false + decode: + nodes: 2 + workers: 1 + env: + TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' + TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP + args: + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + max_batch_size: 16 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + moe_config: + backend: WIDEEP + benchmark: + concurrencies: '308' + +override_disagg_2p1d_dep16_b8_eplb0_mtp3: + name: h100_8k1k_ctx2dep16_gen1dep16_batch8_eplb0_mtp3 + roles: + prefill: + nodes: 4 + workers: 2 + args: + enable_chunked_prefill: true + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 2 + workers: 1 + env: + TRTLLM_DISABLE_KV_CACHE_TRANSFER_OVERLAP: '1' + TRTLLM_FORCE_ALLTOALL_METHOD: DeepEP + args: + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + max_batch_size: 8 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + moe_config: + backend: WIDEEP + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '154' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p1d-dep8-b256-eplb0-mtp0-c128.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p1d-dep8-b256-eplb0-mtp0-c128.yaml deleted file mode 100644 index 6d29cf84b7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p1d-dep8-b256-eplb0-mtp0-c128.yaml +++ /dev/null @@ -1,118 +0,0 @@ -schema: 2 -name: "c128_ctx1_gen1_dep8_batch256_eplb0_mtp0" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (DEP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - # Matches E2E standalone ctx_config.yaml - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - - decode: - nodes: 1 - - workers: 1 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (DEP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - # Matches E2E standalone gen_config.yaml (DEP c=128) - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: true - enable_chunked_prefill: false - max_batch_size: 256 - max_num_tokens: 256 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128, 256] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "128" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p1d-dep8-b32-eplb0-mtp2-c64.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p1d-dep8-b32-eplb0-mtp2-c64.yaml deleted file mode 100644 index 678260d324..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p1d-dep8-b32-eplb0-mtp2-c64.yaml +++ /dev/null @@ -1,121 +0,0 @@ -schema: 2 -name: "c64_ctx1_gen1_dep8_batch32_eplb0_mtp2" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (MTP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 2 - decode: - nodes: 1 - - workers: 1 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (MTP c=64) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: true - enable_chunked_prefill: false - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 2 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "64" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p3d-tep8-b32-eplb0-mtp0-c48.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p3d-tep8-b32-eplb0-mtp0-c48.yaml deleted file mode 100644 index 0ffaa88578..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p3d-tep8-b32-eplb0-mtp0-c48.yaml +++ /dev/null @@ -1,115 +0,0 @@ -schema: 2 -name: "c16_ctx1_gen3_tep8_batch32_eplb0_mtp0" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (TEP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - decode: - nodes: 3 - - workers: 3 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (TEP c=16) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 32 - max_num_tokens: 32 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "48" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p3d-tep8-b32-eplb0-mtp2-c48.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p3d-tep8-b32-eplb0-mtp2-c48.yaml deleted file mode 100644 index 159a19aa6a..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p3d-tep8-b32-eplb0-mtp2-c48.yaml +++ /dev/null @@ -1,121 +0,0 @@ -schema: 2 -name: "c16_ctx1_gen3_tep8_batch32_eplb0_mtp2" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (MTP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 2 - decode: - nodes: 3 - - workers: 3 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (MTP c=16) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 2 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "48" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p6d-tep8-b16-eplb0-mtp0-c48.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p6d-tep8-b16-eplb0-mtp0-c48.yaml deleted file mode 100644 index 26b57c7336..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p6d-tep8-b16-eplb0-mtp0-c48.yaml +++ /dev/null @@ -1,115 +0,0 @@ -schema: 2 -name: "c8_ctx1_gen6_tep8_batch16_eplb0_mtp0" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (TEP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - decode: - nodes: 6 - - workers: 6 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (TEP c=8) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 16 - max_num_tokens: 16 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "48" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p6d-tep8-b32-eplb0-mtp3-c48.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p6d-tep8-b32-eplb0-mtp3-c48.yaml deleted file mode 100644 index 68c9cecf6b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p6d-tep8-b32-eplb0-mtp3-c48.yaml +++ /dev/null @@ -1,121 +0,0 @@ -schema: 2 -name: "c8_ctx1_gen6_tep8_batch32_eplb0_mtp3" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (MTP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - decode: - nodes: 6 - - workers: 6 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (MTP c=8) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "48" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b1-eplb0-mtp0-c9.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b1-eplb0-mtp0-c9.yaml deleted file mode 100644 index f4acc636d5..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b1-eplb0-mtp0-c9.yaml +++ /dev/null @@ -1,115 +0,0 @@ -schema: 2 -name: "c1_ctx1_gen7_tep8_batch1_eplb0_mtp0" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (TEP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - decode: - nodes: 7 - - workers: 7 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (TEP c=4) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 1 - max_num_tokens: 1 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "9" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b1-eplb0-mtp3-c9.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b1-eplb0-mtp3-c9.yaml deleted file mode 100644 index bc547bb206..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b1-eplb0-mtp3-c9.yaml +++ /dev/null @@ -1,121 +0,0 @@ -schema: 2 -name: "c1_ctx1_gen7_tep8_batch1_eplb0_mtp3" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (MTP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - decode: - nodes: 7 - - workers: 7 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (MTP c=4) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 1 - max_num_tokens: 4 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "9" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b32-eplb0-mtp0-c28.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b32-eplb0-mtp0-c28.yaml deleted file mode 100644 index a16cedac66..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b32-eplb0-mtp0-c28.yaml +++ /dev/null @@ -1,115 +0,0 @@ -schema: 2 -name: "c4_ctx1_gen7_tep8_batch32_eplb0_mtp0" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (TEP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - decode: - nodes: 7 - - workers: 7 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (TEP c=4) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 32 - max_num_tokens: 32 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "28" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b32-eplb0-mtp3-c28.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b32-eplb0-mtp3-c28.yaml deleted file mode 100644 index 58b602d0a0..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b32-eplb0-mtp3-c28.yaml +++ /dev/null @@ -1,121 +0,0 @@ -schema: 2 -name: "c4_ctx1_gen7_tep8_batch32_eplb0_mtp3" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (MTP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - decode: - nodes: 7 - - workers: 7 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (MTP c=4) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "28" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p1d-dep8-b32-eplb0-mtp2-c128.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p1d-dep8-b32-eplb0-mtp2-c128.yaml deleted file mode 100644 index ff2f750024..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p1d-dep8-b32-eplb0-mtp2-c128.yaml +++ /dev/null @@ -1,121 +0,0 @@ -schema: 2 -name: "c128_ctx2_gen1_dep8_batch32_eplb0_mtp2" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 2 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (MTP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 2 - decode: - nodes: 1 - - workers: 1 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (MTP c=128) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: true - enable_chunked_prefill: false - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 2 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "128" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p3d-dep8-b128-eplb0-mtp0-c192.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p3d-dep8-b128-eplb0-mtp0-c192.yaml deleted file mode 100644 index d84d12d166..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p3d-dep8-b128-eplb0-mtp0-c192.yaml +++ /dev/null @@ -1,115 +0,0 @@ -schema: 2 -name: "c64_ctx2_gen3_dep8_batch128_eplb0_mtp0" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 2 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (DEP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - decode: - nodes: 3 - - workers: 3 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (DEP c=64) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: true - enable_chunked_prefill: false - max_batch_size: 128 - max_num_tokens: 128 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "192" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p5d-tep8-b128-eplb0-mtp0-c160.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p5d-tep8-b128-eplb0-mtp0-c160.yaml deleted file mode 100644 index 69ff3f63b3..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p5d-tep8-b128-eplb0-mtp0-c160.yaml +++ /dev/null @@ -1,115 +0,0 @@ -schema: 2 -name: "c32_ctx2_gen5_tep8_batch128_eplb0_mtp0" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 2 - workers: 2 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (TEP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - decode: - nodes: 5 - - workers: 5 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (TEP c=32) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 128 - max_num_tokens: 128 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "160" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b32-eplb0-mtp2-c256.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b32-eplb0-mtp2-c256.yaml deleted file mode 100644 index a984eaa2c8..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b32-eplb0-mtp2-c256.yaml +++ /dev/null @@ -1,121 +0,0 @@ -schema: 2 -name: "c256_ctx3_gen1_dep8_batch32_eplb0_mtp2" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 3 - workers: 3 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (MTP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 2 - decode: - nodes: 1 - - workers: 1 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (MTP c=256) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: true - enable_chunked_prefill: false - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 2 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "256" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b512-eplb0-mtp0-c512.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b512-eplb0-mtp0-c512.yaml deleted file mode 100644 index 750fec9cac..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b512-eplb0-mtp0-c512.yaml +++ /dev/null @@ -1,115 +0,0 @@ -schema: 2 -name: "c512_ctx3_gen1_dep8_batch512_eplb0_mtp0" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 3 - workers: 3 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (DEP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - decode: - nodes: 1 - - workers: 1 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (DEP c=512) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: true - enable_chunked_prefill: false - max_batch_size: 512 - max_num_tokens: 512 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128, 256, 512] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "512" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp1-c512.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp1-c512.yaml deleted file mode 100644 index 8ccf739fe4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp1-c512.yaml +++ /dev/null @@ -1,121 +0,0 @@ -schema: 2 -name: "c512_ctx3_gen1_dep8_batch64_eplb0_mtp1" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 3 - workers: 3 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (MTP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - decode: - nodes: 1 - - workers: 1 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (MTP c=512) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: true - enable_chunked_prefill: false - max_batch_size: 64 - max_num_tokens: 256 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32, 64] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 1 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "512" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p5d-tep8-b32-eplb0-mtp3-c160.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p5d-tep8-b32-eplb0-mtp3-c160.yaml deleted file mode 100644 index d9063d91f7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p5d-tep8-b32-eplb0-mtp3-c160.yaml +++ /dev/null @@ -1,121 +0,0 @@ -schema: 2 -name: "c32_ctx3_gen5_tep8_batch32_eplb0_mtp3" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 3 - workers: 3 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (MTP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - decode: - nodes: 5 - - workers: 5 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (MTP c=32) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - speculative_config: - decoding_type: MTP - num_nextn_predict_layers: 3 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "160" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-5p3d-dep8-b256-eplb0-mtp0-c768.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-5p3d-dep8-b256-eplb0-mtp0-c768.yaml deleted file mode 100644 index 5e5d015dd8..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-5p3d-dep8-b256-eplb0-mtp0-c768.yaml +++ /dev/null @@ -1,115 +0,0 @@ -schema: 2 -name: "c256_ctx5_gen3_dep8_batch256_eplb0_mtp0" - -model: - path: "dsr1" - container: "nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1" - precision: "fp8" - -sbatch_directives: - cpus-per-gpu: "16" - -resources: - gpu_type: "h200" - gpus_per_node: 8 - -engine: trtllm -roles: - prefill: - nodes: 5 - workers: 5 - - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Prefill Worker Config for Dynamo DSR1 (DEP mode) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: false - enable_chunked_prefill: false - max_batch_size: 2 - max_num_tokens: 16640 - max_seq_len: 8232 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 32768 - moe_config: - backend: CUTLASS - cuda_graph_config: - disable_overlap_scheduler: true - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - decode: - nodes: 3 - - workers: 3 - env: - UCX_TLS: "rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp" - TRTLLM_ENABLE_PDL: "1" - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - - args: - # Decode Worker Config for Dynamo DSR1 (DEP c=256) - # ISL/OSL: 8k/1k, TP=8 on H200 - backend: pytorch - trust_remote_code: true - tensor_parallel_size: 8 - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - enable_attention_dp: true - enable_chunked_prefill: false - max_batch_size: 256 - max_num_tokens: 256 - max_seq_len: 9256 - kv_cache_config: - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - dtype: fp8 - cache_transceiver_config: - backend: UCX - max_tokens_in_buffer: 16384 - moe_config: - backend: CUTLASS - use_low_precision_moe_combine: true - cuda_graph_config: - enable_padding: true - batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128, 256] - disable_overlap_scheduler: false - print_iter_log: true - # Performance tuning - stream_interval: 100 - num_postprocess_workers: 4 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "768" - req_rate: "inf" - -frontend: - type: "dynamo" - enable_multiple_frontends: false # For some reason, the H200 cluster doesn't like nginx. - -dynamo: - install: false - # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. - # Use NATS for a recipe prior to Dynamo commit 39d2a68. - request_plane: "nats" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml new file mode 100644 index 0000000000..0edc7ff744 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml @@ -0,0 +1,524 @@ +# srt-slurm recipes for dsr1/trtllm/h200-fp8/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_. + +schema: 2 + +base: + model: + path: dsr1 + container: nvcr.io#nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1 + precision: fp8 + sbatch_directives: + cpus-per-gpu: '16' + resources: + gpu_type: h200 + gpus_per_node: 8 + engine: trtllm + roles: + prefill: + env: + UCX_TLS: rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp + TRTLLM_ENABLE_PDL: '1' + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + args: + # ISL/OSL: 8k/1k, TP=8 on H200 + backend: pytorch + trust_remote_code: true + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + pipeline_parallel_size: 1 + enable_attention_dp: false + enable_chunked_prefill: false + max_batch_size: 2 + max_num_tokens: 16640 + max_seq_len: 8232 + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + backend: UCX + max_tokens_in_buffer: 32768 + moe_config: + backend: CUTLASS + cuda_graph_config: null + disable_overlap_scheduler: true + print_iter_log: true + # Performance tuning + stream_interval: 100 + num_postprocess_workers: 4 + decode: + env: + UCX_TLS: rc,dc,ud,cuda_copy,cuda_ipc,gdr_copy,tcp + TRTLLM_ENABLE_PDL: '1' + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + args: + # ISL/OSL: 8k/1k, TP=8 on H200 + backend: pytorch + trust_remote_code: true + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + pipeline_parallel_size: 1 + enable_chunked_prefill: false + max_seq_len: 9256 + kv_cache_config: + enable_block_reuse: false + free_gpu_memory_fraction: 0.85 + dtype: fp8 + cache_transceiver_config: + backend: UCX + max_tokens_in_buffer: 16384 + moe_config: + backend: CUTLASS + use_low_precision_moe_combine: true + cuda_graph_config: + enable_padding: true + disable_overlap_scheduler: false + print_iter_log: true + # Performance tuning + stream_interval: 100 + num_postprocess_workers: 4 + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + frontend: + type: dynamo + # For some reason, the H200 cluster doesn't like nginx. + enable_multiple_frontends: false + dynamo: + install: false + # The container tag identifies a pre-39d2a68 Dynamo release; the image is not digest-pinned. + # Use NATS for a recipe prior to Dynamo commit 39d2a68. + request_plane: nats + +override_disagg_1p1d_dep8_b256_eplb0_mtp0_c128: + name: c128_ctx1_gen1_dep8_batch256_eplb0_mtp0 + roles: + # Prefill Worker Config for Dynamo DSR1 (DEP mode) + # Matches E2E standalone ctx_config.yaml + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 1 + workers: 1 + # Decode Worker Config for Dynamo DSR1 (DEP mode) + # Matches E2E standalone gen_config.yaml (DEP c=128) + args: + enable_attention_dp: true + max_batch_size: 256 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128, 256] + benchmark: + concurrencies: '128' + +override_disagg_1p1d_dep8_b32_eplb0_mtp2_c64: + name: c64_ctx1_gen1_dep8_batch32_eplb0_mtp2 + roles: + prefill: + nodes: 1 + workers: 1 + # Prefill Worker Config for Dynamo DSR1 (MTP mode) + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 2 + decode: + nodes: 1 + workers: 1 + # Decode Worker Config for Dynamo DSR1 (MTP c=64) + args: + enable_attention_dp: true + max_batch_size: 32 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32] + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 2 + benchmark: + concurrencies: '64' + +override_disagg_1p3d_tep8_b32_eplb0_mtp0_c48: + name: c16_ctx1_gen3_tep8_batch32_eplb0_mtp0 + roles: + # Prefill Worker Config for Dynamo DSR1 (TEP mode) + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 3 + workers: 3 + # Decode Worker Config for Dynamo DSR1 (TEP c=16) + args: + enable_attention_dp: false + max_batch_size: 32 + max_num_tokens: 32 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32] + benchmark: + concurrencies: '48' + +override_disagg_1p3d_tep8_b32_eplb0_mtp2_c48: + name: c16_ctx1_gen3_tep8_batch32_eplb0_mtp2 + roles: + prefill: + nodes: 1 + workers: 1 + # Prefill Worker Config for Dynamo DSR1 (MTP mode) + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 2 + decode: + nodes: 3 + workers: 3 + # Decode Worker Config for Dynamo DSR1 (MTP c=16) + args: + enable_attention_dp: false + max_batch_size: 32 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32] + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 2 + benchmark: + concurrencies: '48' + +override_disagg_1p6d_tep8_b16_eplb0_mtp0_c48: + name: c8_ctx1_gen6_tep8_batch16_eplb0_mtp0 + roles: + # Prefill Worker Config for Dynamo DSR1 (TEP mode) + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 6 + workers: 6 + # Decode Worker Config for Dynamo DSR1 (TEP c=8) + args: + enable_attention_dp: false + max_batch_size: 16 + max_num_tokens: 16 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + benchmark: + concurrencies: '48' + +override_disagg_1p6d_tep8_b32_eplb0_mtp3_c48: + name: c8_ctx1_gen6_tep8_batch32_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + # Prefill Worker Config for Dynamo DSR1 (MTP mode) + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 6 + workers: 6 + # Decode Worker Config for Dynamo DSR1 (MTP c=8) + args: + enable_attention_dp: false + max_batch_size: 32 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32] + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '48' + +override_disagg_1p7d_tep8_b1_eplb0_mtp0_c9: + name: c1_ctx1_gen7_tep8_batch1_eplb0_mtp0 + roles: + # Prefill Worker Config for Dynamo DSR1 (TEP mode) + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 7 + workers: 7 + # Decode Worker Config for Dynamo DSR1 (TEP c=4) + args: + enable_attention_dp: false + max_batch_size: 1 + max_num_tokens: 1 + cuda_graph_config: + batch_sizes: [1] + benchmark: + concurrencies: '9' + +override_disagg_1p7d_tep8_b1_eplb0_mtp3_c9: + name: c1_ctx1_gen7_tep8_batch1_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + # Prefill Worker Config for Dynamo DSR1 (MTP mode) + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 7 + workers: 7 + # Decode Worker Config for Dynamo DSR1 (MTP c=4) + args: + enable_attention_dp: false + max_batch_size: 1 + max_num_tokens: 4 + cuda_graph_config: + batch_sizes: [1] + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '9' + +override_disagg_1p7d_tep8_b32_eplb0_mtp0_c28: + name: c4_ctx1_gen7_tep8_batch32_eplb0_mtp0 + roles: + # Prefill Worker Config for Dynamo DSR1 (TEP mode) + prefill: + nodes: 1 + workers: 1 + decode: + nodes: 7 + workers: 7 + # Decode Worker Config for Dynamo DSR1 (TEP c=4) + args: + enable_attention_dp: false + max_batch_size: 32 + max_num_tokens: 32 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32] + benchmark: + concurrencies: '28' + +override_disagg_1p7d_tep8_b32_eplb0_mtp3_c28: + name: c4_ctx1_gen7_tep8_batch32_eplb0_mtp3 + roles: + prefill: + nodes: 1 + workers: 1 + # Prefill Worker Config for Dynamo DSR1 (MTP mode) + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 7 + workers: 7 + # Decode Worker Config for Dynamo DSR1 (MTP c=4) + args: + enable_attention_dp: false + max_batch_size: 32 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32] + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '28' + +override_disagg_2p1d_dep8_b32_eplb0_mtp2_c128: + name: c128_ctx2_gen1_dep8_batch32_eplb0_mtp2 + roles: + prefill: + nodes: 2 + workers: 2 + # Prefill Worker Config for Dynamo DSR1 (MTP mode) + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 2 + decode: + nodes: 1 + workers: 1 + # Decode Worker Config for Dynamo DSR1 (MTP c=128) + args: + enable_attention_dp: true + max_batch_size: 32 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32] + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 2 + benchmark: + concurrencies: '128' + +override_disagg_2p3d_dep8_b128_eplb0_mtp0_c192: + name: c64_ctx2_gen3_dep8_batch128_eplb0_mtp0 + roles: + # Prefill Worker Config for Dynamo DSR1 (DEP mode) + prefill: + nodes: 2 + workers: 2 + decode: + nodes: 3 + workers: 3 + # Decode Worker Config for Dynamo DSR1 (DEP c=64) + args: + enable_attention_dp: true + max_batch_size: 128 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128] + benchmark: + concurrencies: '192' + +override_disagg_2p5d_tep8_b128_eplb0_mtp0_c160: + name: c32_ctx2_gen5_tep8_batch128_eplb0_mtp0 + roles: + # Prefill Worker Config for Dynamo DSR1 (TEP mode) + prefill: + nodes: 2 + workers: 2 + decode: + nodes: 5 + workers: 5 + # Decode Worker Config for Dynamo DSR1 (TEP c=32) + args: + enable_attention_dp: false + max_batch_size: 128 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128] + benchmark: + concurrencies: '160' + +override_disagg_3p1d_dep8_b32_eplb0_mtp2_c256: + name: c256_ctx3_gen1_dep8_batch32_eplb0_mtp2 + roles: + prefill: + nodes: 3 + workers: 3 + # Prefill Worker Config for Dynamo DSR1 (MTP mode) + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 2 + decode: + nodes: 1 + workers: 1 + # Decode Worker Config for Dynamo DSR1 (MTP c=256) + args: + enable_attention_dp: true + max_batch_size: 32 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32] + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 2 + benchmark: + concurrencies: '256' + +override_disagg_3p1d_dep8_b512_eplb0_mtp0_c512: + name: c512_ctx3_gen1_dep8_batch512_eplb0_mtp0 + roles: + # Prefill Worker Config for Dynamo DSR1 (DEP mode) + prefill: + nodes: 3 + workers: 3 + decode: + nodes: 1 + workers: 1 + # Decode Worker Config for Dynamo DSR1 (DEP c=512) + args: + enable_attention_dp: true + max_batch_size: 512 + max_num_tokens: 512 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128, 256, 512] + benchmark: + concurrencies: '512' + +override_disagg_3p1d_dep8_b64_eplb0_mtp1_c512: + name: c512_ctx3_gen1_dep8_batch64_eplb0_mtp1 + roles: + prefill: + nodes: 3 + workers: 3 + # Prefill Worker Config for Dynamo DSR1 (MTP mode) + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + decode: + nodes: 1 + workers: 1 + # Decode Worker Config for Dynamo DSR1 (MTP c=512) + args: + enable_attention_dp: true + max_batch_size: 64 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64] + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 1 + benchmark: + concurrencies: '512' + +override_disagg_3p5d_tep8_b32_eplb0_mtp3_c160: + name: c32_ctx3_gen5_tep8_batch32_eplb0_mtp3 + roles: + prefill: + nodes: 3 + workers: 3 + # Prefill Worker Config for Dynamo DSR1 (MTP mode) + args: + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + decode: + nodes: 5 + workers: 5 + # Decode Worker Config for Dynamo DSR1 (MTP c=32) + args: + enable_attention_dp: false + max_batch_size: 32 + max_num_tokens: 128 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32] + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + benchmark: + concurrencies: '160' + +override_disagg_5p3d_dep8_b256_eplb0_mtp0_c768: + name: c256_ctx5_gen3_dep8_batch256_eplb0_mtp0 + roles: + # Prefill Worker Config for Dynamo DSR1 (DEP mode) + prefill: + nodes: 5 + workers: 5 + decode: + nodes: 3 + workers: 3 + # Decode Worker Config for Dynamo DSR1 (DEP c=256) + args: + enable_attention_dp: true + max_batch_size: 256 + max_num_tokens: 256 + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128, 256] + benchmark: + concurrencies: '768' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c1-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c1-mtp-hicache.yaml deleted file mode 100644 index f2368fa8a8..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c1-mtp-hicache.yaml +++ /dev/null @@ -1,136 +0,0 @@ -schema: 2 -name: "agg-b200-tp8-c1-mtp-hicache" - -# B200 AgentX aggregate topology: one TP8 worker uses one eight-GPU node -# and serves both prefill and decode with bundled DSpark and HiCache. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" - frameworks: - dynamo: "1.5.0.dev20260914" - -dynamo: - install: true - source: - wheel: "1.5.0.dev20260914" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: "b200" - gpus_per_node: 8 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: false - env: - PIP_BREAK_SYSTEM_PACKAGES: "1" - DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 8 - - env: - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - SGLANG_DEFAULT_THINKING: "1" - SGLANG_DSV4_REASONING_EFFORT: high - PIP_BREAK_SYSTEM_PACKAGES: "1" - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" - SGLANG_OPT_USE_ONLINE_COMPRESS: "0" - SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1" - SGLANG_OPT_USE_JIT_NORM: "1" - SGLANG_OPT_USE_TOPK_V2: "True" - - args: - served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813" - enable-metrics: true - enable-cache-report: true - trust-remote-code: true - weight-loader-prefetch-checkpoints: true - stream-interval: 10 - watchdog-timeout: 1000000 - mem-fraction-static: 0.90 - page-size: 256 - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - moe-runner-backend: "flashinfer_mxfp4" - enable-deepseek-v4-fp4-indexer: true - disable-flashinfer-autotune: true - swa-full-tokens-ratio: 0.1 - max-running-requests: 2 - cuda-graph-max-bs-decode: 2 - scheduler-recv-interval: 30 - dp-size: 1 - tp-size: 8 - ep-size: 1 - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-ratio: 2.75 - hicache-write-policy: write_through - hicache-io-backend: direct - hicache-mem-layout: page_first_direct - -sbatch_directives: - mem: "0" - # NScale B200 nodes expose 192 logical CPUs. Request the full node CPU set; - # mem=0 reserves all 1.7 TiB of allocatable host DRAM for HiCache. - cpus-per-task: "192" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "false" - TP: "8" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c4-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c4-mtp-hicache.yaml deleted file mode 100644 index 5cc7a371c4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c4-mtp-hicache.yaml +++ /dev/null @@ -1,136 +0,0 @@ -schema: 2 -name: "agg-b200-tp8-c4-mtp-hicache" - -# B200 AgentX aggregate topology: one TP8 worker uses one eight-GPU node -# and serves both prefill and decode with bundled DSpark and HiCache. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" - frameworks: - dynamo: "1.5.0.dev20260914" - -dynamo: - install: true - source: - wheel: "1.5.0.dev20260914" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: "b200" - gpus_per_node: 8 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: false - env: - PIP_BREAK_SYSTEM_PACKAGES: "1" - DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 8 - - env: - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - SGLANG_DEFAULT_THINKING: "1" - SGLANG_DSV4_REASONING_EFFORT: high - PIP_BREAK_SYSTEM_PACKAGES: "1" - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" - SGLANG_OPT_USE_ONLINE_COMPRESS: "0" - SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1" - SGLANG_OPT_USE_JIT_NORM: "1" - SGLANG_OPT_USE_TOPK_V2: "True" - - args: - served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813" - enable-metrics: true - enable-cache-report: true - trust-remote-code: true - weight-loader-prefetch-checkpoints: true - stream-interval: 10 - watchdog-timeout: 1000000 - mem-fraction-static: 0.90 - page-size: 256 - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - moe-runner-backend: "flashinfer_mxfp4" - enable-deepseek-v4-fp4-indexer: true - disable-flashinfer-autotune: true - swa-full-tokens-ratio: 0.1 - max-running-requests: 8 - cuda-graph-max-bs-decode: 8 - scheduler-recv-interval: 30 - dp-size: 1 - tp-size: 8 - ep-size: 1 - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-ratio: 2.75 - hicache-write-policy: write_through - hicache-io-backend: direct - hicache-mem-layout: page_first_direct - -sbatch_directives: - mem: "0" - # NScale B200 nodes expose 192 logical CPUs. Request the full node CPU set; - # mem=0 reserves all 1.7 TiB of allocatable host DRAM for HiCache. - cpus-per-task: "192" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "false" - TP: "8" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c8-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c8-mtp-hicache.yaml deleted file mode 100644 index e1f28fa87d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c8-mtp-hicache.yaml +++ /dev/null @@ -1,136 +0,0 @@ -schema: 2 -name: "agg-b200-tp8-c8-mtp-hicache" - -# B200 AgentX aggregate topology: one TP8 worker uses one eight-GPU node -# and serves both prefill and decode with bundled DSpark and HiCache. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" - frameworks: - dynamo: "1.5.0.dev20260914" - -dynamo: - install: true - source: - wheel: "1.5.0.dev20260914" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: "b200" - gpus_per_node: 8 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: false - env: - PIP_BREAK_SYSTEM_PACKAGES: "1" - DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 8 - - env: - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - SGLANG_DEFAULT_THINKING: "1" - SGLANG_DSV4_REASONING_EFFORT: high - PIP_BREAK_SYSTEM_PACKAGES: "1" - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" - SGLANG_OPT_USE_ONLINE_COMPRESS: "0" - SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1" - SGLANG_OPT_USE_JIT_NORM: "1" - SGLANG_OPT_USE_TOPK_V2: "True" - - args: - served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813" - enable-metrics: true - enable-cache-report: true - trust-remote-code: true - weight-loader-prefetch-checkpoints: true - stream-interval: 10 - watchdog-timeout: 1000000 - mem-fraction-static: 0.90 - page-size: 256 - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - moe-runner-backend: "flashinfer_mxfp4" - enable-deepseek-v4-fp4-indexer: true - disable-flashinfer-autotune: true - swa-full-tokens-ratio: 0.1 - max-running-requests: 16 - cuda-graph-max-bs-decode: 16 - scheduler-recv-interval: 30 - dp-size: 1 - tp-size: 8 - ep-size: 1 - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-ratio: 2.75 - hicache-write-policy: write_through - hicache-io-backend: direct - hicache-mem-layout: page_first_direct - -sbatch_directives: - mem: "0" - # NScale B200 nodes expose 192 logical CPUs. Request the full node CPU set; - # mem=0 reserves all 1.7 TiB of allocatable host DRAM for HiCache. - cpus-per-task: "192" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "false" - TP: "8" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-1p1d-dep8-dep8-c128-mtp-kvoffload.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-1p1d-dep8-dep8-c128-mtp-kvoffload.yaml deleted file mode 100644 index 305bf5d7ef..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-1p1d-dep8-dep8-c128-mtp-kvoffload.yaml +++ /dev/null @@ -1,234 +0,0 @@ -schema: 2 -name: "disagg-b200-1p1d-dep8-dep8-c128-mtp-kvoffload" - -# B200 AgentX SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 -# (1P x DEP8 / 1D x DEP8, bundled DSpark + HiCache KV offload), tuned for concurrency 128. -# Each DEP8 worker occupies one eight-GPU B200 node. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" - frameworks: - dynamo: "1.5.0.dev20260914" - -dynamo: - install: true - source: - wheel: "1.5.0.dev20260914" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: b200 - gpus_per_node: 8 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: "1" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 8 - env: - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - SGLANG_DSV4_MHC_PREWARM: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' - SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - # P/D workers are on separate HGX B200 nodes, so use Mooncake's standard - # RDMA-registered CUDA buffers rather than an NVL72 custom memory pool. - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - # Use the eight external HCAs. mlx5_6..9 are on a separate low-LID - # fabric, while mlx5_bond_0 caused cross-fabric QP RTR timeouts. - disaggregation-ib-device: mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_10,mlx5_11 - disaggregation-mode: prefill - load-balance-method: total_tokens - mem-fraction-static: 0.85 - page-size: 256 - # Favor the full-attention pool; measured SWA utilization remained low. - swa-full-tokens-ratio: 0.01 - max-running-requests: 256 - cuda-graph-max-bs-decode: 256 - chunked-prefill-size: 65536 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-ratio: 8 - hicache-io-backend: direct - hicache-mem-layout: page_first_direct - - decode: - nodes: 1 - workers: 1 - gpus: 8 - - env: - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_RAGGED_VERIFY_MODE: "static" - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - disaggregation-ib-device: mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_10,mlx5_11 - disaggregation-mode: decode - load-balance-method: total_tokens - mem-fraction-static: 0.9 - page-size: 256 - swa-full-tokens-ratio: 0.02 - max-running-requests: 256 - cuda-graph-max-bs-decode: 256 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - - -sbatch_directives: - mem: "0" - # NScale B200 nodes expose 192 logical CPUs. Request the full node CPU set; - # mem=0 reserves all 1.7 TiB of allocatable host DRAM for HiCache. - cpus-per-task: "192" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml deleted file mode 100644 index 16f1ed6c8e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml +++ /dev/null @@ -1,234 +0,0 @@ -schema: 2 -name: "disagg-b200-1p1d-dep8-dep8-c64-mtp-kvoffload" - -# B200 AgentX SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 -# (1P x DEP8 / 1D x DEP8, bundled DSpark + HiCache KV offload), tuned for concurrency 64. -# Each DEP8 worker occupies one eight-GPU B200 node. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" - frameworks: - dynamo: "1.5.0.dev20260914" - -dynamo: - install: true - source: - wheel: "1.5.0.dev20260914" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: b200 - gpus_per_node: 8 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: "1" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 8 - env: - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - SGLANG_DSV4_MHC_PREWARM: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' - SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - # P/D workers are on separate HGX B200 nodes, so use Mooncake's standard - # RDMA-registered CUDA buffers rather than an NVL72 custom memory pool. - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - # Use the eight external HCAs. mlx5_6..9 are on a separate low-LID - # fabric, while mlx5_bond_0 caused cross-fabric QP RTR timeouts. - disaggregation-ib-device: mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_10,mlx5_11 - disaggregation-mode: prefill - load-balance-method: total_tokens - mem-fraction-static: 0.85 - page-size: 256 - # Favor the full-attention pool; measured SWA utilization remained low. - swa-full-tokens-ratio: 0.01 - max-running-requests: 256 - cuda-graph-max-bs-decode: 256 - chunked-prefill-size: 65536 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-ratio: 8 - hicache-io-backend: direct - hicache-mem-layout: page_first_direct - - decode: - nodes: 1 - workers: 1 - gpus: 8 - - env: - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_RAGGED_VERIFY_MODE: "static" - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - disaggregation-ib-device: mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_10,mlx5_11 - disaggregation-mode: decode - load-balance-method: total_tokens - mem-fraction-static: 0.9 - page-size: 256 - swa-full-tokens-ratio: 0.02 - max-running-requests: 128 - cuda-graph-max-bs-decode: 256 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - - -sbatch_directives: - mem: "0" - # NScale B200 nodes expose 192 logical CPUs. Request the full node CPU set; - # mem=0 reserves all 1.7 TiB of allocatable host DRAM for HiCache. - cpus-per-task: "192" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-2p1d-dep8-dep8-c256-mtp-kvoffload.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-2p1d-dep8-dep8-c256-mtp-kvoffload.yaml deleted file mode 100644 index 4ac6c19f63..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-2p1d-dep8-dep8-c256-mtp-kvoffload.yaml +++ /dev/null @@ -1,239 +0,0 @@ -schema: 2 -name: "disagg-b200-2p1d-dep8-dep8-c256-mtp-kvoffload" - -# B200 AgentX SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 -# (2P x DEP8 / 1D x DEP8, bundled DSpark + HiCache KV offload), tuned for concurrency 256. -# Each DEP8 worker occupies one eight-GPU B200 node. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" - frameworks: - dynamo: "1.5.0.dev20260914" - -dynamo: - install: true - source: - wheel: "1.5.0.dev20260914" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: b200 - gpus_per_node: 8 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: "1" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - prefill: - nodes: 2 - workers: 2 - gpus: 8 - env: - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' - SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - # P/D workers are on separate HGX B200 nodes, so use Mooncake's standard - # RDMA-registered CUDA buffers rather than an NVL72 custom memory pool. - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - SGLANG_DSV4_MHC_PREWARM: '1' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - # Use the eight external HCAs. mlx5_6..9 are on a separate low-LID - # fabric, while mlx5_bond_0 caused cross-fabric QP RTR timeouts. - disaggregation-ib-device: mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_10,mlx5_11 - disaggregation-mode: prefill - load-balance-method: total_tokens - mem-fraction-static: 0.85 - page-size: 256 - swa-full-tokens-ratio: 0.02 - max-running-requests: 256 - cuda-graph-max-bs-decode: 256 - chunked-prefill-size: 65536 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-ratio: 8 - hicache-io-backend: direct - hicache-mem-layout: page_first_direct - - decode: - nodes: 1 - workers: 1 - gpus: 8 - - - env: - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_RAGGED_VERIFY_MODE: "static" - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - SGLANG_DSV4_MHC_PREWARM: '1' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - disaggregation-ib-device: mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_10,mlx5_11 - disaggregation-mode: decode - load-balance-method: total_tokens - # Leave enough decode activation headroom while expanding the KV pools - # for transient DP-rank imbalance at concurrency 256. - mem-fraction-static: 0.91 - page-size: 256 - # Favor the full-attention pool while retaining enough SWA capacity for - # the busiest decode rank. - swa-full-tokens-ratio: 0.005 - max-running-requests: 512 - cuda-graph-max-bs-decode: 256 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - - -sbatch_directives: - mem: "0" - # NScale B200 nodes expose 192 logical CPUs. Request the full node CPU set; - # mem=0 reserves all 1.7 TiB of allocatable host DRAM for HiCache. - cpus-per-task: "192" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..812be2d069 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml @@ -0,0 +1,798 @@ +# srt-slurm recipes for dsv4/sglang/b200-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: deepseek-v4-pro-0813 + container: dynamo-sglang + precision: fp4 + identity: + model: + repo: deepseek-ai/DeepSeek-V4-Pro-0813 + container: + image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 + frameworks: + dynamo: 1.5.0.dev20260914 + dynamo: + install: true + source: + wheel: 1.5.0.dev20260914 + health_check: + max_attempts: 1440 + interval_seconds: 10 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + engine: sglang + roles: {} + sbatch_directives: + mem: '0' + # NScale B200 nodes expose 192 logical CPUs. Request the full node CPU set; + # mem=0 reserves all 1.7 TiB of allocatable host DRAM for HiCache. + cpus-per-task: '192' + srun_options: + mem: '0' + container-remap-root: '' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + +# (model) B200 AgentX aggregate topology: one TP8 worker uses one eight-GPU node +# (model) and serves both prefill and decode with bundled DSpark and HiCache. +override_agg_b200_tp8_c1_mtp_hicache: + name: agg-b200-tp8-c1-mtp-hicache + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + enable_multiple_frontends: false + env: + DYN_NATS_REQUEST_TIMEOUT_SECS: '1800' + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + PIP_BREAK_SYSTEM_PACKAGES: '1' + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + SGLANG_OPT_USE_JIT_INDEXER_METADATA: '1' + SGLANG_OPT_USE_JIT_NORM: '1' + SGLANG_OPT_USE_TOPK_V2: 'True' + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + trust-remote-code: true + weight-loader-prefetch-checkpoints: true + stream-interval: 10 + watchdog-timeout: 1000000 + mem-fraction-static: 0.9 + page-size: 256 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + moe-runner-backend: flashinfer_mxfp4 + enable-deepseek-v4-fp4-indexer: true + disable-flashinfer-autotune: true + swa-full-tokens-ratio: 0.1 + max-running-requests: 2 + cuda-graph-max-bs-decode: 2 + scheduler-recv-interval: 30 + dp-size: 1 + tp-size: 8 + ep-size: 1 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-ratio: 2.75 + hicache-write-policy: write_through + hicache-io-backend: direct + hicache-mem-layout: page_first_direct + benchmark: + env: + IS_MULTINODE: 'false' + TP: '8' + +# (model) B200 AgentX aggregate topology: one TP8 worker uses one eight-GPU node +# (model) and serves both prefill and decode with bundled DSpark and HiCache. +override_agg_b200_tp8_c4_mtp_hicache: + name: agg-b200-tp8-c4-mtp-hicache + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + enable_multiple_frontends: false + env: + DYN_NATS_REQUEST_TIMEOUT_SECS: '1800' + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + PIP_BREAK_SYSTEM_PACKAGES: '1' + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + SGLANG_OPT_USE_JIT_INDEXER_METADATA: '1' + SGLANG_OPT_USE_JIT_NORM: '1' + SGLANG_OPT_USE_TOPK_V2: 'True' + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + trust-remote-code: true + weight-loader-prefetch-checkpoints: true + stream-interval: 10 + watchdog-timeout: 1000000 + mem-fraction-static: 0.9 + page-size: 256 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + moe-runner-backend: flashinfer_mxfp4 + enable-deepseek-v4-fp4-indexer: true + disable-flashinfer-autotune: true + swa-full-tokens-ratio: 0.1 + max-running-requests: 8 + cuda-graph-max-bs-decode: 8 + scheduler-recv-interval: 30 + dp-size: 1 + tp-size: 8 + ep-size: 1 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-ratio: 2.75 + hicache-write-policy: write_through + hicache-io-backend: direct + hicache-mem-layout: page_first_direct + benchmark: + env: + IS_MULTINODE: 'false' + TP: '8' + +# (model) B200 AgentX aggregate topology: one TP8 worker uses one eight-GPU node +# (model) and serves both prefill and decode with bundled DSpark and HiCache. +override_agg_b200_tp8_c8_mtp_hicache: + name: agg-b200-tp8-c8-mtp-hicache + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + enable_multiple_frontends: false + env: + DYN_NATS_REQUEST_TIMEOUT_SECS: '1800' + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + PIP_BREAK_SYSTEM_PACKAGES: '1' + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + SGLANG_OPT_USE_JIT_INDEXER_METADATA: '1' + SGLANG_OPT_USE_JIT_NORM: '1' + SGLANG_OPT_USE_TOPK_V2: 'True' + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + trust-remote-code: true + weight-loader-prefetch-checkpoints: true + stream-interval: 10 + watchdog-timeout: 1000000 + mem-fraction-static: 0.9 + page-size: 256 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + moe-runner-backend: flashinfer_mxfp4 + enable-deepseek-v4-fp4-indexer: true + disable-flashinfer-autotune: true + swa-full-tokens-ratio: 0.1 + max-running-requests: 16 + cuda-graph-max-bs-decode: 16 + scheduler-recv-interval: 30 + dp-size: 1 + tp-size: 8 + ep-size: 1 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-ratio: 2.75 + hicache-write-policy: write_through + hicache-io-backend: direct + hicache-mem-layout: page_first_direct + benchmark: + env: + IS_MULTINODE: 'false' + TP: '8' + +# (model) B200 AgentX SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 +# (model) (1P x DEP8 / 1D x DEP8, bundled DSpark + HiCache KV offload), tuned for concurrency 128. +# (model) Each DEP8 worker occupies one eight-GPU B200 node. +override_disagg_b200_1p1d_dep8_dep8_c128_mtp_kvoffload: + name: disagg-b200-1p1d-dep8-dep8-c128-mtp-kvoffload + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 + frontend: + enable_multiple_frontends: true + num_additional_frontends: 4 + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '900' + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: '1' + SGLANG_DSV4_MHC_PREWARM: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + # P/D workers are on separate HGX B200 nodes, so use Mooncake's standard + # RDMA-registered CUDA buffers rather than an NVL72 custom memory pool. + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + # Use the eight external HCAs. mlx5_6..9 are on a separate low-LID + # fabric, while mlx5_bond_0 caused cross-fabric QP RTR timeouts. + disaggregation-ib-device: mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_10,mlx5_11 + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + # Favor the full-attention pool; measured SWA utilization remained low. + swa-full-tokens-ratio: 0.01 + max-running-requests: 256 + cuda-graph-max-bs-decode: 256 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 8 + hicache-io-backend: direct + hicache-mem-layout: page_first_direct + decode: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '900' + SGLANG_RAGGED_VERIFY_MODE: static + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-ib-device: mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_10,mlx5_11 + disaggregation-mode: decode + load-balance-method: total_tokens + mem-fraction-static: 0.9 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 256 + cuda-graph-max-bs-decode: 256 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + benchmark: + env: + IS_MULTINODE: 'true' + +# (model) B200 AgentX SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 +# (model) Each DEP8 worker occupies one eight-GPU B200 node. +# (model) (1P x DEP8 / 1D x DEP8, bundled DSpark + HiCache KV offload), tuned for concurrency 64. +override_disagg_b200_1p1d_dep8_dep8_c64_mtp_kvoffload: + name: disagg-b200-1p1d-dep8-dep8-c64-mtp-kvoffload + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 + frontend: + enable_multiple_frontends: true + num_additional_frontends: 4 + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '900' + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: '1' + SGLANG_DSV4_MHC_PREWARM: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + # P/D workers are on separate HGX B200 nodes, so use Mooncake's standard + # RDMA-registered CUDA buffers rather than an NVL72 custom memory pool. + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + # Use the eight external HCAs. mlx5_6..9 are on a separate low-LID + # fabric, while mlx5_bond_0 caused cross-fabric QP RTR timeouts. + disaggregation-ib-device: mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_10,mlx5_11 + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + # Favor the full-attention pool; measured SWA utilization remained low. + swa-full-tokens-ratio: 0.01 + max-running-requests: 256 + cuda-graph-max-bs-decode: 256 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 8 + hicache-io-backend: direct + hicache-mem-layout: page_first_direct + decode: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '900' + SGLANG_RAGGED_VERIFY_MODE: static + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-ib-device: mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_10,mlx5_11 + disaggregation-mode: decode + load-balance-method: total_tokens + mem-fraction-static: 0.9 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 128 + cuda-graph-max-bs-decode: 256 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + benchmark: + env: + IS_MULTINODE: 'true' + +# (model) B200 AgentX SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 +# (model) Each DEP8 worker occupies one eight-GPU B200 node. +# (model) (2P x DEP8 / 1D x DEP8, bundled DSpark + HiCache KV offload), tuned for concurrency 256. +override_disagg_b200_2p1d_dep8_dep8_c256_mtp_kvoffload: + name: disagg-b200-2p1d-dep8-dep8-c256-mtp-kvoffload + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 + frontend: + enable_multiple_frontends: true + num_additional_frontends: 4 + roles: + prefill: + nodes: 2 + workers: 2 + gpus: 8 + env: + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '900' + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + # P/D workers are on separate HGX B200 nodes, so use Mooncake's standard + # RDMA-registered CUDA buffers rather than an NVL72 custom memory pool. + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_DSV4_MHC_PREWARM: '1' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + # Use the eight external HCAs. mlx5_6..9 are on a separate low-LID + # fabric, while mlx5_bond_0 caused cross-fabric QP RTR timeouts. + disaggregation-ib-device: mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_10,mlx5_11 + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 256 + cuda-graph-max-bs-decode: 256 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 8 + hicache-io-backend: direct + hicache-mem-layout: page_first_direct + decode: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '900' + SGLANG_RAGGED_VERIFY_MODE: static + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_DSV4_MHC_PREWARM: '1' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-ib-device: mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_10,mlx5_11 + disaggregation-mode: decode + load-balance-method: total_tokens + # Leave enough decode activation headroom while expanding the KV pools + # for transient DP-rank imbalance at concurrency 256. + mem-fraction-static: 0.91 + page-size: 256 + # Favor the full-attention pool while retaining enough SWA capacity for + # the busiest decode rank. + swa-full-tokens-ratio: 0.005 + max-running-requests: 512 + cuda-graph-max-bs-decode: 256 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + benchmark: + env: + IS_MULTINODE: 'true' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-tp4-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-tp4-mtp.yaml deleted file mode 100644 index c49541b92b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-tp4-mtp.yaml +++ /dev/null @@ -1,130 +0,0 @@ -schema: 2 -name: "agg-gb300-tp4-mtp-lowlatency" - -# Low-latency AgentX aggregate topology: one TP4 worker occupies one -# four-GPU GB300 node and serves both prefill and decode with DSpark K=6. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:v0.5.19-cu130@sha256:d6e7288627be8b02be88e4bba38e73f6d50e2826869f753c13a4c4385ab3eda9" - -dynamo: - install: true - source: - wheel: "1.5.0.dev20260910" - -slurm: - time_limit: "4:00:00" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: false - env: - PIP_BREAK_SYSTEM_PACKAGES: "1" - DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - - env: - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_DEFAULT_THINKING: "1" - SGLANG_DSV4_REASONING_EFFORT: high - PIP_BREAK_SYSTEM_PACKAGES: "1" - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" - SGLANG_OPT_USE_ONLINE_COMPRESS: "0" - SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1" - SGLANG_OPT_USE_JIT_NORM: "1" - SGLANG_OPT_USE_TOPK_V2: "True" - - args: - served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813" - enable-metrics: true - enable-cache-report: true - trust-remote-code: true - weight-loader-prefetch-checkpoints: true - stream-interval: 10 - watchdog-timeout: 1000000 - mem-fraction-static: 0.94 - page-size: 256 - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - moe-runner-backend: "flashinfer_mxfp4" - enable-deepseek-v4-fp4-indexer: true - disable-flashinfer-autotune: true - swa-full-tokens-ratio: 0.1 - max-running-requests: 32 - cuda-graph-max-bs-decode: 32 - scheduler-recv-interval: 30 - dp-size: 1 - tp-size: 4 - ep-size: 1 - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - -sbatch_directives: - mem: "0" - cpus-per-task: "144" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "false" - TP: "4" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-tp8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-tp8-mtp.yaml deleted file mode 100644 index 0aea65fbef..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-tp8-mtp.yaml +++ /dev/null @@ -1,130 +0,0 @@ -schema: 2 -name: "agg-gb300-tp8-mtp-lowlatency" - -# Low-latency AgentX aggregate topology: one TP8 worker spans two -# four-GPU GB300 nodes and serves both prefill and decode with DSpark K=6. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:v0.5.19-cu130@sha256:d6e7288627be8b02be88e4bba38e73f6d50e2826869f753c13a4c4385ab3eda9" - -dynamo: - install: true - source: - wheel: "1.5.0.dev20260910" - -slurm: - time_limit: "4:00:00" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: false - env: - PIP_BREAK_SYSTEM_PACKAGES: "1" - DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - - env: - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - SGLANG_DEFAULT_THINKING: "1" - SGLANG_DSV4_REASONING_EFFORT: high - PIP_BREAK_SYSTEM_PACKAGES: "1" - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" - SGLANG_OPT_USE_ONLINE_COMPRESS: "0" - SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1" - SGLANG_OPT_USE_JIT_NORM: "1" - SGLANG_OPT_USE_TOPK_V2: "True" - - args: - served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813" - enable-metrics: true - enable-cache-report: true - trust-remote-code: true - weight-loader-prefetch-checkpoints: true - stream-interval: 10 - watchdog-timeout: 1000000 - mem-fraction-static: 0.94 - page-size: 256 - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - moe-runner-backend: "flashinfer_mxfp4" - enable-deepseek-v4-fp4-indexer: true - disable-flashinfer-autotune: true - swa-full-tokens-ratio: 0.1 - max-running-requests: 4 - cuda-graph-max-bs-decode: 4 - scheduler-recv-interval: 30 - dp-size: 1 - tp-size: 8 - ep-size: 1 - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - -sbatch_directives: - mem: "0" - cpus-per-task: "144" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "false" - TP: "8" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-1p1d-dep8-dep16-c480-mtp-kvoffload.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-1p1d-dep8-dep16-c480-mtp-kvoffload.yaml deleted file mode 100644 index 15885fb9af..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-1p1d-dep8-dep16-c480-mtp-kvoffload.yaml +++ /dev/null @@ -1,234 +0,0 @@ -schema: 2 -name: "disagg-gb300-2p4d-dep8-dep16-c480-mtp-kvoffload" - -# Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 -# (2P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 480. -# -# Uses the flat single-variant srtctl schema the agentic CI flow expects; -# resources + backend (prefill/decode env + sglang_config) are normalized -# from the Pareto run. -# Concurrency is exported into agentic_srt.sh from the master-config conc-list. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21" - -dynamo: - install: true - source: - wheel: "1.5.0.dev20260902" - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: gb300 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: "1" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - env: - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_DSV4_MHC_PREWARM: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' - SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - MC_FORCE_MNNVL: '1' - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - disaggregation-mode: prefill - load-balance-method: total_tokens - mem-fraction-static: 0.85 - page-size: 256 - swa-full-tokens-ratio: 0.02 - max-running-requests: 256 - cuda-graph-max-bs: 256 - chunked-prefill-size: 65536 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-ratio: 1 - hicache-io-backend: direct - - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - MC_FORCE_MNNVL: '1' - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 16 - dp-size: 16 - ep-size: 16 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - disaggregation-mode: decode - load-balance-method: total_tokens - mem-fraction-static: 0.9 - page-size: 256 - swa-full-tokens-ratio: 0.02 - max-running-requests: 3072 - cuda-graph-max-bs: 256 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - - -sbatch_directives: - mem: "0" - cpus-per-task: "144" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-2p1d-dep8-dep16-c960-mtp-kvoffload.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-2p1d-dep8-dep16-c960-mtp-kvoffload.yaml deleted file mode 100644 index 00ce318b0d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-2p1d-dep8-dep16-c960-mtp-kvoffload.yaml +++ /dev/null @@ -1,236 +0,0 @@ -schema: 2 -name: "disagg-gb300-4p4d-dep8-dep16-c960-mtp-kvoffload" - -# Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro on GB300 -# (4P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 960. -# -# DEP8-prefill variant aligned with the measured Pareto point: prefill uses -# tp/dp/ep 8, four nodes, and SGLANG_DSV4_MHC_PREWARM=1. -# Concurrency is exported into agentic_srt.sh -# from the master-config conc-list. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21" - -dynamo: - install: true - source: - wheel: "1.5.0.dev20260902" - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: gb300 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: "1" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - prefill: - nodes: 4 - workers: 2 - env: - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - OMP_NUM_THREADS: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' - SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - MC_FORCE_MNNVL: '1' - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - SGLANG_DSV4_MHC_PREWARM: '1' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - disaggregation-mode: prefill - load-balance-method: total_tokens - mem-fraction-static: 0.85 - page-size: 256 - swa-full-tokens-ratio: 0.02 - max-running-requests: 256 - cuda-graph-max-bs: 256 - chunked-prefill-size: 65536 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-ratio: 1 - hicache-io-backend: direct - - decode: - nodes: 4 - workers: 1 - - - env: - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - OMP_NUM_THREADS: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - MC_FORCE_MNNVL: '1' - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - SGLANG_DSV4_MHC_PREWARM: '1' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 16 - dp-size: 16 - ep-size: 16 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - disaggregation-mode: decode - load-balance-method: total_tokens - mem-fraction-static: 0.9 - page-size: 256 - swa-full-tokens-ratio: 0.02 - max-running-requests: 3072 - cuda-graph-max-bs: 256 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - - -sbatch_directives: - mem: "0" - cpus-per-task: "144" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-3p1d-dep8-dep16-c1440-mtp-kvoffload.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-3p1d-dep8-dep16-c1440-mtp-kvoffload.yaml deleted file mode 100644 index f0e66a32d8..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-3p1d-dep8-dep16-c1440-mtp-kvoffload.yaml +++ /dev/null @@ -1,234 +0,0 @@ -schema: 2 -name: "disagg-gb300-6p4d-dep8-dep16-c1440-mtp-kvoffload" - -# Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 -# (6P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1440. -# -# Uses the flat single-variant srtctl schema the agentic CI flow expects; -# resources + backend (prefill/decode env + sglang_config) are normalized -# from the Pareto run. -# Concurrency is exported into agentic_srt.sh from the master-config conc-list. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21" - -dynamo: - install: true - source: - wheel: "1.5.0.dev20260902" - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: gb300 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: "1" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - prefill: - nodes: 6 - workers: 3 - gpus: 8 - env: - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_DSV4_MHC_PREWARM: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' - SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - MC_FORCE_MNNVL: '1' - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - disaggregation-mode: prefill - load-balance-method: total_tokens - mem-fraction-static: 0.85 - page-size: 256 - swa-full-tokens-ratio: 0.02 - max-running-requests: 512 - cuda-graph-max-bs: 512 - chunked-prefill-size: 65536 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-ratio: 1 - hicache-io-backend: direct - - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - MC_FORCE_MNNVL: '1' - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 16 - dp-size: 16 - ep-size: 16 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - disaggregation-mode: decode - load-balance-method: total_tokens - mem-fraction-static: 0.9 - page-size: 256 - swa-full-tokens-ratio: 0.02 - max-running-requests: 3072 - cuda-graph-max-bs: 512 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - - -sbatch_directives: - mem: "0" - cpus-per-task: "144" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-4p1d-dep8-dep16-c1920-mtp-kvoffload.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-4p1d-dep8-dep16-c1920-mtp-kvoffload.yaml deleted file mode 100644 index e569bd41cd..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-4p1d-dep8-dep16-c1920-mtp-kvoffload.yaml +++ /dev/null @@ -1,239 +0,0 @@ -schema: 2 -name: "disagg-gb300-8p4d-dep8-dep16-c1920-mtp-kvoffload" - -# Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 -# (8P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1920. -# -# Uses the flat single-variant srtctl schema the agentic CI flow expects; -# resources + backend (prefill/decode env + sglang_config) are normalized -# from the Pareto run. -# Concurrency is exported into agentic_srt.sh from the master-config conc-list. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21" - -dynamo: - install: true - source: - wheel: "1.5.0.dev20260902" - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: gb300 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - nginx_keepalive_timeout: "900s" - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - # AgentX warmup can legitimately keep the single wide decode worker busy - # for longer than Dynamo's 10-second TCP request-plane default. - DYN_TCP_REQUEST_TIMEOUT: "60" - PIP_BREAK_SYSTEM_PACKAGES: "1" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - prefill: - nodes: 8 - workers: 4 - gpus: 8 - env: - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600" - SGLANG_DSV4_MHC_PREWARM: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' - SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - MC_FORCE_MNNVL: '1' - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - disaggregation-mode: prefill - load-balance-method: total_tokens - mem-fraction-static: 0.85 - page-size: 256 - swa-full-tokens-ratio: 0.02 - max-running-requests: 1024 - cuda-graph-max-bs: 1024 - chunked-prefill-size: 65536 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-ratio: 1 - hicache-io-backend: direct - - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600" - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - MC_FORCE_MNNVL: '1' - NCCL_TIMEOUT: '100000' - NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 16 - dp-size: 16 - ep-size: 16 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - disaggregation-mode: decode - load-balance-method: total_tokens - mem-fraction-static: 0.9 - page-size: 256 - swa-full-tokens-ratio: 0.02 - max-running-requests: 3072 - cuda-graph-max-bs: 192 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - - -sbatch_directives: - mem: "0" - cpus-per-task: "144" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..059b675d01 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml @@ -0,0 +1,932 @@ +# srt-slurm recipes for dsv4/sglang/gb300-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: deepseek-v4-pro-0813 + container: dynamo-sglang + precision: fp4 + identity: + model: + repo: deepseek-ai/DeepSeek-V4-Pro-0813 + container: {} + dynamo: + install: true + source: {} + slurm: {} + health_check: + max_attempts: 1440 + interval_seconds: 10 + resources: + gpu_type: gb300 + gpus_per_node: 4 + frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + engine: sglang + roles: {} + sbatch_directives: + mem: '0' + cpus-per-task: '144' + srun_options: + mem: '0' + container-remap-root: '' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + +# (model) Low-latency AgentX aggregate topology: one TP4 worker occupies one +# (model) four-GPU GB300 node and serves both prefill and decode with DSpark K=6. +override_agg_tp4_mtp: + name: agg-gb300-tp4-mtp-lowlatency + identity: + container: + image: lmsysorg/sglang:v0.5.19-cu130@sha256:d6e7288627be8b02be88e4bba38e73f6d50e2826869f753c13a4c4385ab3eda9 + dynamo: + source: + wheel: 1.5.0.dev20260910 + slurm: + time_limit: '4:00:00' + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + enable_multiple_frontends: false + env: + DYN_NATS_REQUEST_TIMEOUT_SECS: '1800' + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + env: + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + PIP_BREAK_SYSTEM_PACKAGES: '1' + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: '1' + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + SGLANG_OPT_USE_JIT_INDEXER_METADATA: '1' + SGLANG_OPT_USE_JIT_NORM: '1' + SGLANG_OPT_USE_TOPK_V2: 'True' + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + trust-remote-code: true + weight-loader-prefetch-checkpoints: true + stream-interval: 10 + watchdog-timeout: 1000000 + mem-fraction-static: 0.94 + page-size: 256 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + moe-runner-backend: flashinfer_mxfp4 + enable-deepseek-v4-fp4-indexer: true + disable-flashinfer-autotune: true + swa-full-tokens-ratio: 0.1 + max-running-requests: 32 + cuda-graph-max-bs-decode: 32 + scheduler-recv-interval: 30 + dp-size: 1 + tp-size: 4 + ep-size: 1 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + benchmark: + env: + IS_MULTINODE: 'false' + TP: '4' + +# (model) Low-latency AgentX aggregate topology: one TP8 worker spans two +# (model) four-GPU GB300 nodes and serves both prefill and decode with DSpark K=6. +override_agg_tp8_mtp: + name: agg-gb300-tp8-mtp-lowlatency + identity: + container: + image: lmsysorg/sglang:v0.5.19-cu130@sha256:d6e7288627be8b02be88e4bba38e73f6d50e2826869f753c13a4c4385ab3eda9 + dynamo: + source: + wheel: 1.5.0.dev20260910 + slurm: + time_limit: '4:00:00' + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + enable_multiple_frontends: false + env: + DYN_NATS_REQUEST_TIMEOUT_SECS: '1800' + roles: + agg: + nodes: 2 + workers: 1 + gpus: 8 + env: + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + PIP_BREAK_SYSTEM_PACKAGES: '1' + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + SGLANG_OPT_USE_JIT_INDEXER_METADATA: '1' + SGLANG_OPT_USE_JIT_NORM: '1' + SGLANG_OPT_USE_TOPK_V2: 'True' + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + trust-remote-code: true + weight-loader-prefetch-checkpoints: true + stream-interval: 10 + watchdog-timeout: 1000000 + mem-fraction-static: 0.94 + page-size: 256 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + moe-runner-backend: flashinfer_mxfp4 + enable-deepseek-v4-fp4-indexer: true + disable-flashinfer-autotune: true + swa-full-tokens-ratio: 0.1 + max-running-requests: 4 + cuda-graph-max-bs-decode: 4 + scheduler-recv-interval: 30 + dp-size: 1 + tp-size: 8 + ep-size: 1 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + benchmark: + env: + IS_MULTINODE: 'false' + TP: '8' + +# (model) Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 +# (model) (2P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 480. +# (model) +# (model) Uses the flat single-variant srtctl schema the agentic CI flow expects; +# (model) resources + backend (prefill/decode env + sglang_config) are normalized +# (model) from the Pareto run. +# (model) Concurrency is exported into agentic_srt.sh from the master-config conc-list. +override_disagg_1p1d_dep8_dep16_c480_mtp_kvoffload: + name: disagg-gb300-2p4d-dep8-dep16-c480-mtp-kvoffload + identity: + container: + image: lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21 + dynamo: + source: + wheel: 1.5.0.dev20260902 + slurm: + time_limit: '8:00:00' + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 + frontend: + enable_multiple_frontends: true + num_additional_frontends: 4 + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + env: + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: '1' + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '900' + SGLANG_DSV4_MHC_PREWARM: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + MC_FORCE_MNNVL: '1' + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 256 + cuda-graph-max-bs: 256 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 1 + hicache-io-backend: direct + decode: + nodes: 4 + workers: 1 + gpus: 16 + env: + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '900' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + MC_FORCE_MNNVL: '1' + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 16 + dp-size: 16 + ep-size: 16 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: decode + load-balance-method: total_tokens + mem-fraction-static: 0.9 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 3072 + cuda-graph-max-bs: 256 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + benchmark: + env: + IS_MULTINODE: 'true' + +# (model) +# (model) Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro on GB300 +# (model) (4P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 960. +# (model) DEP8-prefill variant aligned with the measured Pareto point: prefill uses +# (model) tp/dp/ep 8, four nodes, and SGLANG_DSV4_MHC_PREWARM=1. +# (model) Concurrency is exported into agentic_srt.sh +# (model) from the master-config conc-list. +override_disagg_2p1d_dep8_dep16_c960_mtp_kvoffload: + name: disagg-gb300-4p4d-dep8-dep16-c960-mtp-kvoffload + identity: + container: + image: lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21 + dynamo: + source: + wheel: 1.5.0.dev20260902 + slurm: + time_limit: '8:00:00' + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 + frontend: + enable_multiple_frontends: true + num_additional_frontends: 4 + roles: + prefill: + nodes: 4 + workers: 2 + env: + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: '1' + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '900' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + OMP_NUM_THREADS: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + MC_FORCE_MNNVL: '1' + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_DSV4_MHC_PREWARM: '1' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 256 + cuda-graph-max-bs: 256 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 1 + hicache-io-backend: direct + decode: + nodes: 4 + workers: 1 + env: + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '900' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + OMP_NUM_THREADS: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + MC_FORCE_MNNVL: '1' + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_DSV4_MHC_PREWARM: '1' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 16 + dp-size: 16 + ep-size: 16 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: decode + load-balance-method: total_tokens + mem-fraction-static: 0.9 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 3072 + cuda-graph-max-bs: 256 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + benchmark: + env: + IS_MULTINODE: 'true' + +# (model) Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 +# (model) +# (model) Uses the flat single-variant srtctl schema the agentic CI flow expects; +# (model) resources + backend (prefill/decode env + sglang_config) are normalized +# (model) from the Pareto run. +# (model) Concurrency is exported into agentic_srt.sh from the master-config conc-list. +# (model) (6P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1440. +override_disagg_3p1d_dep8_dep16_c1440_mtp_kvoffload: + name: disagg-gb300-6p4d-dep8-dep16-c1440-mtp-kvoffload + identity: + container: + image: lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21 + dynamo: + source: + wheel: 1.5.0.dev20260902 + slurm: + time_limit: '8:00:00' + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 + frontend: + enable_multiple_frontends: true + num_additional_frontends: 4 + roles: + prefill: + nodes: 6 + workers: 3 + gpus: 8 + env: + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: '1' + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '900' + SGLANG_DSV4_MHC_PREWARM: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + MC_FORCE_MNNVL: '1' + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 512 + cuda-graph-max-bs: 512 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 1 + hicache-io-backend: direct + decode: + nodes: 4 + workers: 1 + gpus: 16 + env: + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '900' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + MC_FORCE_MNNVL: '1' + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 16 + dp-size: 16 + ep-size: 16 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: decode + load-balance-method: total_tokens + mem-fraction-static: 0.9 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 3072 + cuda-graph-max-bs: 512 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + benchmark: + env: + IS_MULTINODE: 'true' + +# (model) Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 +# (model) +# (model) Uses the flat single-variant srtctl schema the agentic CI flow expects; +# (model) resources + backend (prefill/decode env + sglang_config) are normalized +# (model) from the Pareto run. +# (model) Concurrency is exported into agentic_srt.sh from the master-config conc-list. +# (model) (8P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1920. +override_disagg_4p1d_dep8_dep16_c1920_mtp_kvoffload: + name: disagg-gb300-8p4d-dep8-dep16-c1920-mtp-kvoffload + identity: + container: + image: lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21 + dynamo: + source: + wheel: 1.5.0.dev20260902 + slurm: + time_limit: '8:00:00' + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 + frontend: + enable_multiple_frontends: true + env: + # AgentX warmup can legitimately keep the single wide decode worker busy + # for longer than Dynamo's 10-second TCP request-plane default. + DYN_TCP_REQUEST_TIMEOUT: '60' + num_additional_frontends: 4 + nginx_keepalive_timeout: 900s + roles: + prefill: + nodes: 8 + workers: 4 + gpus: 8 + env: + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: '1' + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '3600' + SGLANG_DSV4_MHC_PREWARM: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + MC_FORCE_MNNVL: '1' + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 1024 + cuda-graph-max-bs: 1024 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 1 + hicache-io-backend: direct + decode: + nodes: 4 + workers: 1 + gpus: 16 + env: + SGLANG_RAGGED_VERIFY_MODE: static + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '3600' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + MC_FORCE_MNNVL: '1' + NCCL_TIMEOUT: '100000' + NCCL_MNNVL_UUID_QUERY_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 16 + dp-size: 16 + ep-size: 16 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: decode + load-balance-method: total_tokens + mem-fraction-static: 0.9 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 3072 + cuda-graph-max-bs: 192 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + benchmark: + env: + IS_MULTINODE: 'true' + AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p1d-dep8-dep32-c388-b4-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p1d-dep8-dep32-c388-b4-mtp.yaml deleted file mode 100644 index adc2d1d3a3..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p1d-dep8-dep32-c388-b4-mtp.yaml +++ /dev/null @@ -1,195 +0,0 @@ -schema: 2 -name: dynamo-disagg-gb300-1p1d-dep8-dep32-c388-b4-mtp -model: - path: deepseek-ai/DeepSeek-V4-Pro - container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 -identity: - model: - repo: deepseek-ai/DeepSeek-V4-Pro - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - request_plane: tcp -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - env: - OMP_NUM_THREADS: '1' - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: '1' - DYN_DEFAULT_THINKING_MODE: disabled - MIMALLOC_ARENA_RESERVE: '0' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - CUDA_SCALE_LAUNCH_QUEUES: 4x - args: - attention_dp_config: - kv_cache_routing_conversation_affinity: true - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - custom_tokenizer: deepseek_v4 - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - host_cache_size: 193273528320 - pool_ratio: - - 0.55 - - 0.22 - - 0.23 - tokens_per_block: 128 - block_reuse_config: - policy: per_conversation - max_num_turns: 5 - max_batch_size: 256 - max_num_tokens: 16384 - max_seq_len: 990016 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 8 - decode: - nodes: 8 - workers: 1 - gpus: 32 - env: - OMP_NUM_THREADS: '1' - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: '1' - DYN_DEFAULT_THINKING_MODE: disabled - MIMALLOC_ARENA_RESERVE: '0' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - custom_tokenizer: deepseek_v4 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - avg_seq_len: 200000 - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.7 - host_cache_size: 0 - tokens_per_block: 128 - max_batch_size: 4 - max_num_tokens: 16 - max_seq_len: 996595 - moe_config: - backend: MEGAMOE_DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 32 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - sparse_attention_config: - algorithm: deepseek_v4 - enable_heuristic_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 20 - tensor_parallel_size: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - MODEL: deepseek-ai/DeepSeek-V4-Pro - MODEL_PREFIX: dsv4 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '388' - DURATION: '3600' - KV_OFFLOADING: dram - ETCD_LEASE_TTL: '120' - DYN_ROUTER_QUEUE_THRESHOLD: None - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TOKENIZER: fastokens - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - placement: - node: first_decode -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - MODEL: deepseek-ai/DeepSeek-V4-Pro - MODEL_PREFIX: dsv4 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '388' - DURATION: '3600' - KV_OFFLOADING: dram - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - SERVED_MODEL_NAME: DeepSeek-V4-Pro - placement: - node: last_decode diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p4d-dep4-tep8-c4-b1-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p4d-dep4-tep8-c4-b1-mtp.yaml deleted file mode 100644 index f4f16337f4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p4d-dep4-tep8-c4-b1-mtp.yaml +++ /dev/null @@ -1,195 +0,0 @@ -schema: 2 -name: dynamo-disagg-gb300-1p4d-dep4-tep8-c4-b1-mtp -model: - path: deepseek-ai/DeepSeek-V4-Pro - container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 -identity: - model: - repo: deepseek-ai/DeepSeek-V4-Pro - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - request_plane: tcp -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - OMP_NUM_THREADS: '1' - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: '1' - DYN_DEFAULT_THINKING_MODE: disabled - MIMALLOC_ARENA_RESERVE: '0' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - CUDA_SCALE_LAUNCH_QUEUES: 4x - args: - attention_dp_config: - kv_cache_routing_conversation_affinity: true - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - custom_tokenizer: deepseek_v4 - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - host_cache_size: 193273528320 - pool_ratio: - - 0.55 - - 0.22 - - 0.23 - tokens_per_block: 128 - block_reuse_config: - policy: per_conversation - max_num_turns: 5 - max_batch_size: 128 - max_num_tokens: 4096 - max_seq_len: 990016 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 4 - decode: - nodes: 8 - workers: 4 - gpus: 8 - env: - OMP_NUM_THREADS: '1' - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: '1' - DYN_DEFAULT_THINKING_MODE: disabled - MIMALLOC_ARENA_RESERVE: '0' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - custom_tokenizer: deepseek_v4 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - avg_seq_len: 200000 - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.9 - host_cache_size: 0 - tokens_per_block: 128 - max_batch_size: 1 - max_num_tokens: 4 - max_seq_len: 996595 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - sparse_attention_config: - algorithm: deepseek_v4 - enable_heuristic_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 20 - tensor_parallel_size: 8 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - MODEL: deepseek-ai/DeepSeek-V4-Pro - MODEL_PREFIX: dsv4 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '4' - DURATION: '3600' - KV_OFFLOADING: dram - ETCD_LEASE_TTL: '120' - DYN_ROUTER_QUEUE_THRESHOLD: None - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TOKENIZER: fastokens - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - placement: - node: first_decode -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - MODEL: deepseek-ai/DeepSeek-V4-Pro - MODEL_PREFIX: dsv4 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '4' - DURATION: '3600' - KV_OFFLOADING: dram - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - SERVED_MODEL_NAME: DeepSeek-V4-Pro - placement: - node: last_decode diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p6d-dep4-tep4-c24-b4-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p6d-dep4-tep4-c24-b4-mtp.yaml deleted file mode 100644 index 08b68b2202..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p6d-dep4-tep4-c24-b4-mtp.yaml +++ /dev/null @@ -1,195 +0,0 @@ -schema: 2 -name: dynamo-disagg-gb300-1p6d-dep4-tep4-c24-b4-mtp -model: - path: deepseek-ai/DeepSeek-V4-Pro - container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 -identity: - model: - repo: deepseek-ai/DeepSeek-V4-Pro - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - request_plane: tcp -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - OMP_NUM_THREADS: '1' - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: '1' - DYN_DEFAULT_THINKING_MODE: disabled - MIMALLOC_ARENA_RESERVE: '0' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - CUDA_SCALE_LAUNCH_QUEUES: 4x - args: - attention_dp_config: - kv_cache_routing_conversation_affinity: true - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - custom_tokenizer: deepseek_v4 - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - host_cache_size: 193273528320 - pool_ratio: - - 0.55 - - 0.22 - - 0.23 - tokens_per_block: 128 - block_reuse_config: - policy: per_conversation - max_num_turns: 5 - max_batch_size: 128 - max_num_tokens: 4096 - max_seq_len: 990016 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 4 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 4 - decode: - nodes: 6 - workers: 6 - gpus: 4 - env: - OMP_NUM_THREADS: '1' - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: '1' - DYN_DEFAULT_THINKING_MODE: disabled - MIMALLOC_ARENA_RESERVE: '0' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - custom_tokenizer: deepseek_v4 - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - avg_seq_len: 200000 - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.9 - host_cache_size: 0 - tokens_per_block: 128 - max_batch_size: 4 - max_num_tokens: 16 - max_seq_len: 996595 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - sparse_attention_config: - algorithm: deepseek_v4 - enable_heuristic_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 20 - tensor_parallel_size: 4 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - MODEL: deepseek-ai/DeepSeek-V4-Pro - MODEL_PREFIX: dsv4 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '24' - DURATION: '3600' - KV_OFFLOADING: dram - ETCD_LEASE_TTL: '120' - DYN_ROUTER_QUEUE_THRESHOLD: None - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TOKENIZER: fastokens - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - placement: - node: first_decode -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - MODEL: deepseek-ai/DeepSeek-V4-Pro - MODEL_PREFIX: dsv4 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '24' - DURATION: '3600' - KV_OFFLOADING: dram - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - SERVED_MODEL_NAME: DeepSeek-V4-Pro - placement: - node: last_decode diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep32-c736-b8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep32-c736-b8-mtp.yaml deleted file mode 100644 index 46be52c30f..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep32-c736-b8-mtp.yaml +++ /dev/null @@ -1,196 +0,0 @@ -schema: 2 -name: dynamo-disagg-gb300-2p1d-dep8-dep32-c736-b8-mtp -model: - path: deepseek-ai/DeepSeek-V4-Pro - container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 -identity: - model: - repo: deepseek-ai/DeepSeek-V4-Pro - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - request_plane: tcp -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false -roles: - prefill: - nodes: 4 - workers: 2 - gpus: 8 - env: - OMP_NUM_THREADS: '1' - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: '1' - DYN_DEFAULT_THINKING_MODE: disabled - MIMALLOC_ARENA_RESERVE: '0' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - CUDA_SCALE_LAUNCH_QUEUES: 4x - args: - attention_dp_config: - kv_cache_routing_conversation_affinity: true - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - custom_tokenizer: deepseek_v4 - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - host_cache_size: 193273528320 - pool_ratio: - - 0.55 - - 0.22 - - 0.23 - tokens_per_block: 128 - block_reuse_config: - policy: per_conversation - max_num_turns: 5 - max_batch_size: 256 - max_num_tokens: 16384 - max_seq_len: 990016 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 8 - decode: - nodes: 8 - workers: 1 - gpus: 32 - env: - OMP_NUM_THREADS: '1' - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: '1' - DYN_DEFAULT_THINKING_MODE: disabled - MIMALLOC_ARENA_RESERVE: '0' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - enable_padding: true - custom_tokenizer: deepseek_v4 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - avg_seq_len: 200000 - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.7 - host_cache_size: 0 - tokens_per_block: 128 - max_batch_size: 8 - max_num_tokens: 32 - max_seq_len: 996595 - moe_config: - backend: MEGAMOE_DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 32 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - sparse_attention_config: - algorithm: deepseek_v4 - enable_heuristic_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 20 - tensor_parallel_size: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - MODEL: deepseek-ai/DeepSeek-V4-Pro - MODEL_PREFIX: dsv4 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '736' - DURATION: '3600' - KV_OFFLOADING: dram - ETCD_LEASE_TTL: '120' - DYN_ROUTER_QUEUE_THRESHOLD: None - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TOKENIZER: fastokens - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - placement: - node: first_decode -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - MODEL: deepseek-ai/DeepSeek-V4-Pro - MODEL_PREFIX: dsv4 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '736' - DURATION: '3600' - KV_OFFLOADING: dram - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - SERVED_MODEL_NAME: DeepSeek-V4-Pro - placement: - node: last_decode diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-3p1d-dep8-dep16-c1152-b32-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-3p1d-dep8-dep16-c1152-b32-mtp.yaml deleted file mode 100644 index 373a12989c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-3p1d-dep8-dep16-c1152-b32-mtp.yaml +++ /dev/null @@ -1,202 +0,0 @@ -schema: 2 -name: dynamo-disagg-gb300-3p1d-dep8-dep16-c1152-b32-mtp -model: - path: deepseek-ai/DeepSeek-V4-Pro - container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 -identity: - model: - repo: deepseek-ai/DeepSeek-V4-Pro - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - request_plane: tcp -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false -roles: - prefill: - nodes: 6 - workers: 3 - gpus: 8 - env: - OMP_NUM_THREADS: '1' - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: '1' - DYN_DEFAULT_THINKING_MODE: disabled - MIMALLOC_ARENA_RESERVE: '0' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - CUDA_SCALE_LAUNCH_QUEUES: 4x - args: - attention_dp_config: - kv_cache_routing_conversation_affinity: true - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - custom_tokenizer: deepseek_v4 - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - host_cache_size: 193273528320 - pool_ratio: - - 0.55 - - 0.22 - - 0.23 - tokens_per_block: 128 - block_reuse_config: - policy: per_conversation - max_num_turns: 5 - max_batch_size: 256 - max_num_tokens: 16384 - max_seq_len: 990016 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 8 - decode: - nodes: 4 - workers: 1 - gpus: 16 - env: - OMP_NUM_THREADS: '1' - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: '1' - DYN_DEFAULT_THINKING_MODE: disabled - MIMALLOC_ARENA_RESERVE: '0' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 12 - - 16 - - 20 - - 24 - - 28 - - 32 - enable_padding: true - custom_tokenizer: deepseek_v4 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - avg_seq_len: 200000 - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.7 - host_cache_size: 0 - tokens_per_block: 128 - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 996595 - moe_config: - backend: MEGAMOE_DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - sparse_attention_config: - algorithm: deepseek_v4 - enable_heuristic_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 20 - tensor_parallel_size: 16 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - MODEL: deepseek-ai/DeepSeek-V4-Pro - MODEL_PREFIX: dsv4 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '1152' - DURATION: '3600' - KV_OFFLOADING: dram - ETCD_LEASE_TTL: '120' - DYN_ROUTER_QUEUE_THRESHOLD: None - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TOKENIZER: fastokens - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - placement: - node: first_decode -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - MODEL: deepseek-ai/DeepSeek-V4-Pro - MODEL_PREFIX: dsv4 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '1152' - DURATION: '3600' - KV_OFFLOADING: dram - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - SERVED_MODEL_NAME: DeepSeek-V4-Pro - placement: - node: last_decode diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-5p1d-dep8-dep16-c2626-b96-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-5p1d-dep8-dep16-c2626-b96-mtp.yaml deleted file mode 100644 index 6e2e980ebd..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-5p1d-dep8-dep16-c2626-b96-mtp.yaml +++ /dev/null @@ -1,218 +0,0 @@ -schema: 2 -name: dynamo-disagg-gb300-5p1d-dep8-dep16-c2626-b96-mtp -model: - path: deepseek-ai/DeepSeek-V4-Pro - container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 -identity: - model: - repo: deepseek-ai/DeepSeek-V4-Pro - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - request_plane: tcp -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false -roles: - prefill: - nodes: 10 - workers: 5 - gpus: 8 - env: - OMP_NUM_THREADS: '1' - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: '1' - DYN_DEFAULT_THINKING_MODE: disabled - MIMALLOC_ARENA_RESERVE: '0' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - CUDA_SCALE_LAUNCH_QUEUES: 4x - args: - attention_dp_config: - kv_cache_routing_conversation_affinity: true - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - custom_tokenizer: deepseek_v4 - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - host_cache_size: 193273528320 - pool_ratio: - - 0.55 - - 0.22 - - 0.23 - tokens_per_block: 128 - block_reuse_config: - policy: per_conversation - max_num_turns: 5 - max_batch_size: 256 - max_num_tokens: 16384 - max_seq_len: 990016 - moe_config: - backend: TRTLLM - moe_expert_parallel_size: 8 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 8 - decode: - nodes: 4 - workers: 1 - gpus: 16 - env: - OMP_NUM_THREADS: '1' - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: '1' - DYN_DEFAULT_THINKING_MODE: disabled - MIMALLOC_ARENA_RESERVE: '0' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 12 - - 16 - - 20 - - 24 - - 28 - - 32 - - 36 - - 40 - - 44 - - 48 - - 52 - - 56 - - 60 - - 64 - - 68 - - 72 - - 76 - - 80 - - 84 - - 88 - - 92 - - 96 - enable_padding: true - custom_tokenizer: deepseek_v4 - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - avg_seq_len: 200000 - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.7 - host_cache_size: 0 - tokens_per_block: 128 - max_batch_size: 96 - max_num_tokens: 384 - max_seq_len: 996595 - moe_config: - backend: MEGAMOE_DEEPGEMM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - sparse_attention_config: - algorithm: deepseek_v4 - enable_heuristic_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 20 - tensor_parallel_size: 16 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - MODEL: deepseek-ai/DeepSeek-V4-Pro - MODEL_PREFIX: dsv4 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '2626' - DURATION: '3600' - KV_OFFLOADING: dram - ETCD_LEASE_TTL: '120' - DYN_ROUTER_QUEUE_THRESHOLD: None - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TOKENIZER: fastokens - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - placement: - node: first_decode -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - MODEL: deepseek-ai/DeepSeek-V4-Pro - MODEL_PREFIX: dsv4 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '2626' - DURATION: '3600' - KV_OFFLOADING: dram - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - SERVED_MODEL_NAME: DeepSeek-V4-Pro - placement: - node: last_decode diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..9ad56d794e --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml @@ -0,0 +1,392 @@ +# srt-slurm recipes for dsv4/trtllm/gb300-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: deepseek-ai/DeepSeek-V4-Pro + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc24 + precision: fp4 + identity: + model: + repo: deepseek-ai/DeepSeek-V4-Pro + container: + image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 + frameworks: + tensorrt_llm: 1.3.0rc24 + dynamo: + install: true + source: + git: https://github.com/cquil11/dynamo.git + rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 + request_plane: tcp + health_check: + max_attempts: 1440 + interval_seconds: 10 + resources: + gpu_type: gb300 + gpus_per_node: 4 + engine: + type: trtllm + publish_events_and_metrics: false + roles: + prefill: + env: + OMP_NUM_THREADS: '1' + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp + DYN_ENGINE_CONV_AFFINITY: '1' + DYN_DEFAULT_THINKING_MODE: disabled + MIMALLOC_ARENA_RESERVE: '0' + TRTLLM_PINNED_WEIGHT_STAGING: '1' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + CUDA_SCALE_LAUNCH_QUEUES: 4x + args: + attention_dp_config: + kv_cache_routing_conversation_affinity: true + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + transceiver_runtime: PYTHON + cuda_graph_config: null + custom_tokenizer: deepseek_v4 + disable_overlap_scheduler: false + enable_attention_dp: true + enable_chunked_prefill: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: true + event_buffer_max_size: 0 + free_gpu_memory_fraction: 0.8 + host_cache_size: 193273528320 + pool_ratio: + - 0.55 + - 0.22 + - 0.23 + tokens_per_block: 128 + block_reuse_config: + policy: per_conversation + max_num_turns: 5 + max_seq_len: 990016 + moe_config: + backend: TRTLLM + pipeline_parallel_size: 1 + print_iter_log: true + enable_iter_perf_stats: true + return_perf_metrics: false + scheduler_config: + capacity_scheduler_policy: MAX_UTILIZATION + speculative_config: + decoding_type: MTP + max_draft_len: 3 + decode: + env: + OMP_NUM_THREADS: '1' + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp + DYN_ENGINE_CONV_AFFINITY: '1' + DYN_DEFAULT_THINKING_MODE: disabled + MIMALLOC_ARENA_RESERVE: '0' + TRTLLM_PINNED_WEIGHT_STAGING: '1' + args: + cache_transceiver_config: + backend: NIXL + kv_transfer_timeout_ms: 600000 + transceiver_runtime: PYTHON + cuda_graph_config: + enable_padding: true + custom_tokenizer: deepseek_v4 + kv_cache_config: + avg_seq_len: 200000 + dtype: fp8 + enable_block_reuse: false + event_buffer_max_size: 0 + host_cache_size: 0 + tokens_per_block: 128 + max_seq_len: 996595 + moe_config: + use_low_precision_moe_combine: true + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + enable_iter_perf_stats: true + return_perf_metrics: false + sparse_attention_config: + algorithm: deepseek_v4 + enable_heuristic_topk: true + speculative_config: + decoding_type: MTP + max_draft_len: 3 + stream_interval: 20 + frontend: + type: dynamo + enable_multiple_frontends: false + env: + MODEL: deepseek-ai/DeepSeek-V4-Pro + MODEL_PREFIX: dsv4 + FRAMEWORK: dynamo-trt + PRECISION: fp4 + DURATION: '3600' + KV_OFFLOADING: dram + ETCD_LEASE_TTL: '120' + DYN_ROUTER_QUEUE_THRESHOLD: None + DYN_TOKENIZER_CACHE: '1' + DYN_TOKENIZER_CACHE_BYTES: '8000000000' + DYN_TOKENIZER: fastokens + DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' + args: + router-mode: kv + no-kv-events: true + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + placement: + node: first_decode + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + MODEL: deepseek-ai/DeepSeek-V4-Pro + MODEL_PREFIX: dsv4 + FRAMEWORK: dynamo-trt + PRECISION: fp4 + DURATION: '3600' + KV_OFFLOADING: dram + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + SERVED_MODEL_NAME: DeepSeek-V4-Pro + placement: + node: last_decode + +override_disagg_1p1d_dep8_dep32_c388_b4_mtp: + name: dynamo-disagg-gb300-1p1d-dep8-dep32-c388-b4-mtp + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + args: + max_batch_size: 256 + max_num_tokens: 16384 + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + decode: + nodes: 8 + workers: 1 + gpus: 32 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 4 + max_num_tokens: 16 + moe_config: + backend: MEGAMOE_DEEPGEMM + moe_expert_parallel_size: 32 + tensor_parallel_size: 32 + frontend: + env: + CONC: '388' + benchmark: + env: + CONC: '388' + +override_disagg_1p4d_dep4_tep8_c4_b1_mtp: + name: dynamo-disagg-gb300-1p4d-dep4-tep8-c4-b1-mtp + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + args: + max_batch_size: 128 + max_num_tokens: 4096 + moe_expert_parallel_size: 4 + tensor_parallel_size: 4 + decode: + nodes: 8 + workers: 4 + gpus: 8 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 1 + max_num_tokens: 4 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + frontend: + env: + CONC: '4' + benchmark: + env: + CONC: '4' + +override_disagg_1p6d_dep4_tep4_c24_b4_mtp: + name: dynamo-disagg-gb300-1p6d-dep4-tep4-c24-b4-mtp + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + args: + max_batch_size: 128 + max_num_tokens: 4096 + moe_expert_parallel_size: 4 + tensor_parallel_size: 4 + decode: + nodes: 6 + workers: 6 + gpus: 4 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 4 + max_num_tokens: 16 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 4 + tensor_parallel_size: 4 + frontend: + env: + CONC: '24' + benchmark: + env: + CONC: '24' + +override_disagg_2p1d_dep8_dep32_c736_b8_mtp: + name: dynamo-disagg-gb300-2p1d-dep8-dep32-c736-b8-mtp + roles: + prefill: + nodes: 4 + workers: 2 + gpus: 8 + args: + max_batch_size: 256 + max_num_tokens: 16384 + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + decode: + nodes: 8 + workers: 1 + gpus: 32 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 8 + max_num_tokens: 32 + moe_config: + backend: MEGAMOE_DEEPGEMM + moe_expert_parallel_size: 32 + tensor_parallel_size: 32 + frontend: + env: + CONC: '736' + benchmark: + env: + CONC: '736' + +override_disagg_3p1d_dep8_dep16_c1152_b32_mtp: + name: dynamo-disagg-gb300-3p1d-dep8-dep16-c1152-b32-mtp + roles: + prefill: + nodes: 6 + workers: 3 + gpus: 8 + args: + max_batch_size: 256 + max_num_tokens: 16384 + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 12, 16, 20, 24, 28, 32] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 32 + max_num_tokens: 128 + moe_config: + backend: MEGAMOE_DEEPGEMM + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + frontend: + env: + CONC: '1152' + benchmark: + env: + CONC: '1152' + +override_disagg_5p1d_dep8_dep16_c2626_b96_mtp: + name: dynamo-disagg-gb300-5p1d-dep8-dep16-c2626-b96-mtp + roles: + prefill: + nodes: 10 + workers: 5 + gpus: 8 + args: + max_batch_size: 256 + max_num_tokens: 16384 + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 12, 16, 20, 24, 28, 32, 36, 40, 44, 48, 52, 56, 60, 64, 68, 72, 76, 80, 84, 88, 92, 96] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 96 + max_num_tokens: 384 + moe_config: + backend: MEGAMOE_DEEPGEMM + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + frontend: + env: + CONC: '2626' + benchmark: + env: + CONC: '2626' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-dep8-mtp.yaml deleted file mode 100644 index 9c604d726b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-dep8-mtp.yaml +++ /dev/null @@ -1,112 +0,0 @@ -schema: 2 -name: dsv4-gb200-vllm-agentic-mtp-agg-dep8 - -model: - path: deepseek-v4-pro - container: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 - precision: fp4 - -identity: - model: {repo: deepseek-ai/DeepSeek-V4-Pro} - container: {image: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106} - frameworks: {dynamo: "1.3.1"} - -dynamo: - install: true - - source: - pypi: "1.3.1" -setup_script: vllm-container-deps.sh -environment: {ETCD_LEASE_TTL: "7200"} -slurm: {time_limit: "8:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} - -resources: - gpu_type: gb200 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: kv - router-reset-states: true - router-session-affinity-ttl-secs: 14400 - kv-cache-block-size: 256 - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - - env: - HF_HUB_CACHE: /hf_hub_cache - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_USE_NCCL_SYMM_MEM: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - args: - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:20080","enable_kv_cache_events":true}' - served-model-name: deepseek-ai/DeepSeek-V4-Pro - kv-cache-dtype: fp8 - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - enable-expert-parallel: true - enable-ep-weight-filter: true - moe-backend: deep_gemm_mega_moe - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - max-model-len: 1048576 - max-num-seqs: 256 - max-num-batched-tokens: 8192 - trust-remote-code: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","custom_ops":["all"]}' - speculative-config: '{"method":"mtp","num_speculative_tokens":2}' - gpu-memory-utilization: 0.90 - stream-interval: 10 - no-disable-hybrid-kv-cache-manager: true - kv-cache-metrics: true - tokenizer-mode: deepseek_v4 - -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-c4-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-c4-mtp3.yaml deleted file mode 100644 index 71f279e59b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-c4-mtp3.yaml +++ /dev/null @@ -1,160 +0,0 @@ -schema: 2 -name: "svf-vllm-agg-gb200-tp8-c4-mtp2-agentic" - -# GB200 AgentX aggregate topology: one TP8 worker spans two four-GPU -# nodes and serves both prefill and decode at concurrency 4. Keep at least -# 16 sequence slots and otherwise size the scheduler at 4x concurrency. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.2.1" - -dynamo: - install: true - - source: - wheel: "1.2.1" -environment: - DYNAMO_WHEEL_DIRS: "/srtctl-wheels" - # The frontend shares Grace CPU capacity with the long TP8 cold start. - ETCD_LEASE_TTL: "7200" - -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "150GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "kv" - router-reset-states: true - router-temperature: 0.0 - router-queue-threshold: 65536 - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - tokenizer: "fastokens" - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "16" - TILELANG_CLEANUP_TEMP_FILES: "1" - VLLM_USE_NCCL_SYMM_MEM: "0" - TORCH_SYMMMEM: "NVSHMEM" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_USE_RUST_FRONTEND: "1" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "auto" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_SPARSE_INDEXER_MAX_LOGITS_MB: "1024" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - NCCL_P2P_LEVEL: "NVL" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb200-agg-mtp2-{job_id}" - args: - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 8 - pipeline-parallel-size: 1 - disable-custom-all-reduce: true - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - max-model-len: 1048576 - max-num-seqs: 16 - max-num-batched-tokens: 8192 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64],"mode":0}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - gpu-memory-utilization: 0.94 - stream-interval: 10 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - -sbatch_directives: - cpus-per-task: "144" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - # Keep aggregate workers in the multinode result schema so ingestion uses - # the zero decode-worker count instead of duplicating TP into P and D. - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-c8-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-c8-mtp3.yaml deleted file mode 100644 index 58e999b60d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-c8-mtp3.yaml +++ /dev/null @@ -1,159 +0,0 @@ -schema: 2 -name: "svf-vllm-agg-gb200-tp8-c8-mtp2-agentic" - -# GB200 AgentX aggregate topology: one TP8 worker spans two four-GPU -# nodes and serves both prefill and decode at concurrency 8. Size max-num-seqs at -# 4x concurrency and expand the MTP CUDA-graph envelope to match. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.2.1" - -dynamo: - install: true - - source: - wheel: "1.2.1" -environment: - DYNAMO_WHEEL_DIRS: "/srtctl-wheels" - ETCD_LEASE_TTL: "7200" - -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "150GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "kv" - router-reset-states: true - router-temperature: 0.0 - router-queue-threshold: 65536 - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - tokenizer: "fastokens" - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "32" - TILELANG_CLEANUP_TEMP_FILES: "1" - VLLM_USE_NCCL_SYMM_MEM: "0" - TORCH_SYMMMEM: "NVSHMEM" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_USE_RUST_FRONTEND: "1" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "auto" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_SPARSE_INDEXER_MAX_LOGITS_MB: "1024" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - NCCL_P2P_LEVEL: "NVL" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb200-agg-tp8-c8-mtp2-{job_id}" - args: - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 8 - pipeline-parallel-size: 1 - disable-custom-all-reduce: true - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - max-model-len: 1048576 - max-num-seqs: 32 - max-num-batched-tokens: 8192 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128],"mode":0}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - gpu-memory-utilization: 0.94 - stream-interval: 10 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - -sbatch_directives: - cpus-per-task: "144" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - # Avoid concurrent readers observing a mismatched mmap data/index pair. - AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-mtp.yaml deleted file mode 100644 index 845cbf966f..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-mtp.yaml +++ /dev/null @@ -1,122 +0,0 @@ -schema: 2 -name: dsv4-gb200-vllm-agentic-mtp-agg-tp8 - -model: - path: deepseek-v4-pro - container: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 - precision: fp4 - -identity: - model: - repo: deepseek-ai/DeepSeek-V4-Pro - container: - image: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 - frameworks: - dynamo: "1.3.1" - -dynamo: - install: true - - source: - pypi: "1.3.1" -setup_script: vllm-container-deps.sh - -environment: - ETCD_LEASE_TTL: "7200" - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: gb200 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: kv - router-reset-states: true - router-session-affinity-ttl-secs: 14400 - kv-cache-block-size: 256 - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - - env: - HF_HUB_CACHE: /hf_hub_cache - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_USE_NCCL_SYMM_MEM: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - args: - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:20080","enable_kv_cache_events":true}' - served-model-name: deepseek-ai/DeepSeek-V4-Pro - kv-cache-dtype: fp8 - tensor-parallel-size: 8 - pipeline-parallel-size: 1 - disable-custom-all-reduce: true - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - max-model-len: 1048576 - max-num-seqs: 128 - max-num-batched-tokens: 8192 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - speculative-config: '{"method":"mtp","num_speculative_tokens":2}' - gpu-memory-utilization: 0.90 - stream-interval: 10 - no-disable-hybrid-kv-cache-manager: true - kv-cache-metrics: true - tokenizer-mode: deepseek_v4 - -sbatch_directives: - cpus-per-task: "144" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-c128-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-c128-mtp3.yaml deleted file mode 100644 index f9c62a6a24..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-c128-mtp3.yaml +++ /dev/null @@ -1,221 +0,0 @@ -schema: 2 -name: "svf-vllm-disagg-gb200-1p1d-dep8-dep8-c128-mtp2-agentic" - -# GB200 AgentX MTP3 topology: one DEP8 prefill worker feeds one -# DEP8 decode worker at concurrency 128. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.3.0.dev20260720" - -dynamo: - install: true - - source: - wheel: "1.3.0.dev20260720" -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "140GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -environment: - # Mooncake prefix-block hashes must match across processes and nodes. - PYTHONHASHSEED: "0" - -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "random" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TCP_CHANNEL_BUFFER: "128" - DYN_TCP_REQUEST_TIMEOUT: "60" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_USE_BREAKABLE_CUDAGRAPH: "0" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb200-1p1d-dep8-dep8-c128-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 8192 - long-prefill-token-threshold: 1024 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - max-cudagraph-capture-size: 256 - gpu-memory-utilization: 0.85 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb200-1p1d-dep8-dep8-c128-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 128 - max-num-batched-tokens: 1024 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - max-cudagraph-capture-size: 512 - gpu-memory-utilization: 0.92 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] -sbatch_directives: - cpus-per-task: "72" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - # Avoid concurrent readers observing a mismatched mmap data/index pair. - AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-c256-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-c256-mtp3.yaml deleted file mode 100644 index 01c6e6e5af..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-c256-mtp3.yaml +++ /dev/null @@ -1,221 +0,0 @@ -schema: 2 -name: "svf-vllm-disagg-gb200-1p1d-dep8-dep8-c256-mtp2-agentic" - -# GB200 AgentX MTP3 topology: one DEP8 prefill worker feeds one -# DEP8 decode worker at concurrency 256. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.3.0.dev20260720" - -dynamo: - install: true - - source: - wheel: "1.3.0.dev20260720" -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "140GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -environment: - # Mooncake prefix-block hashes must match across processes and nodes. - PYTHONHASHSEED: "0" - -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "random" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TCP_CHANNEL_BUFFER: "128" - DYN_TCP_REQUEST_TIMEOUT: "60" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_USE_BREAKABLE_CUDAGRAPH: "0" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb200-1p1d-dep8-dep8-c256-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 8192 - long-prefill-token-threshold: 1024 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - max-cudagraph-capture-size: 256 - gpu-memory-utilization: 0.85 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb200-1p1d-dep8-dep8-c256-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 128 - max-num-batched-tokens: 1024 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - max-cudagraph-capture-size: 512 - gpu-memory-utilization: 0.92 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] -sbatch_directives: - cpus-per-task: "72" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - # Avoid concurrent readers observing a mismatched mmap data/index pair. - AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-mtp.yaml deleted file mode 100644 index 178f6af471..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-mtp.yaml +++ /dev/null @@ -1,133 +0,0 @@ -schema: 2 -name: dsv4-gb200-vllm-agentic-mtp-disagg-1p1d-dep8-dep8 - -model: - path: deepseek-v4-pro - container: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 - precision: fp4 - -identity: - model: {repo: deepseek-ai/DeepSeek-V4-Pro} - container: {image: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106} - frameworks: {dynamo: "1.3.1"} - -dynamo: - install: true - - source: - pypi: "1.3.1" -setup_script: vllm-container-deps.sh -environment: {ETCD_LEASE_TTL: "7200"} -slurm: {time_limit: "8:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} - -resources: - gpu_type: gb200 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - # Long AgentX prefills can exceed Dynamo's request-plane default while - # the healthy DEP8 worker is still computing the first response. - DYN_TCP_REQUEST_TIMEOUT: "60" - args: - router-mode: kv - router-reset-states: true - router-session-affinity-ttl-secs: 14400 - kv-cache-block-size: 256 - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - env: &worker_environment - HF_HUB_CACHE: /hf_hub_cache - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_USE_NCCL_SYMM_MEM: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - args: &dep8_config - kv-transfer-config: '{"kv_connector":"NixlConnector","kv_role":"kv_both"}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:20080","enable_kv_cache_events":true}' - served-model-name: deepseek-ai/DeepSeek-V4-Pro - kv-cache-dtype: fp8 - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - enable-expert-parallel: true - enable-ep-weight-filter: true - moe-backend: deep_gemm_mega_moe - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - max-model-len: 1048576 - max-num-seqs: 16 - max-num-batched-tokens: 16384 - trust-remote-code: true - enforce-eager: true - block-size: 256 - gpu-memory-utilization: 0.95 - no-disable-hybrid-kv-cache-manager: true - kv-cache-metrics: true - enable-sleep-mode: true - tokenizer-mode: deepseek_v4 - speculative-config: '{"method":"mtp","num_speculative_tokens":2}' - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: *worker_environment - args: - <<: *dep8_config - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_fp4_indexer_cache":true}' - enforce-eager: false - gpu-memory-utilization: 0.90 - max-num-seqs: 1024 - max-num-batched-tokens: 1024 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - max-cudagraph-capture-size: 1024 - stream-interval: 10 - -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep12-c576-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep12-c576-mtp3.yaml deleted file mode 100644 index 0bc695d779..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep12-c576-mtp3.yaml +++ /dev/null @@ -1,223 +0,0 @@ -schema: 2 -name: "svf-vllm-disagg-gb200-2p1d-dep8-dep12-c576-mtp2-agentic" - -# GB200 AgentX MTP3 topology: two DEP8 prefill workers feed one -# DEP12 decode worker at concurrency 576. Decode consumes P/D KV through NIXL -# and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.3.0.dev20260720" - -dynamo: - install: true - - source: - wheel: "1.3.0.dev20260720" -environment: - # Mooncake prefix-block hashes must match across processes and nodes. - PYTHONHASHSEED: "0" - -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "140GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "random" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TCP_CHANNEL_BUFFER: "128" - DYN_TCP_REQUEST_TIMEOUT: "60" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 4 - workers: 2 - gpus: 8 - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_MOONCAKE_STORE_SEND_THREADS: "8" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_USE_BREAKABLE_CUDAGRAPH: "0" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb200-2p1d-prefill-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 8192 - long-prefill-token-threshold: 1024 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] - decode: - nodes: 3 - workers: 1 - gpus: 12 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb200-2p1d-dep12-decode-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 12 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 256 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - max-cudagraph-capture-size: 256 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - # Avoid concurrent readers observing a mismatched mmap data/index pair. - AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep16-c512-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep16-c512-mtp3.yaml deleted file mode 100644 index 9ead5965b0..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep16-c512-mtp3.yaml +++ /dev/null @@ -1,223 +0,0 @@ -schema: 2 -name: "svf-vllm-disagg-gb200-2p1d-dep8-dep16-c512-mtp2-agentic" - -# GB200 AgentX MTP3 topology: two DEP8 prefill workers feed one -# DEP16 decode worker at concurrency 512. Decode consumes P/D KV through NIXL -# and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.3.0.dev20260720" - -dynamo: - install: true - - source: - wheel: "1.3.0.dev20260720" -environment: - # Mooncake prefix-block hashes must match across processes and nodes. - PYTHONHASHSEED: "0" - -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "140GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "random" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TCP_CHANNEL_BUFFER: "128" - DYN_TCP_REQUEST_TIMEOUT: "60" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 4 - workers: 2 - gpus: 8 - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_MOONCAKE_STORE_SEND_THREADS: "8" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_USE_BREAKABLE_CUDAGRAPH: "0" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb200-2p1d-prefill-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 8192 - long-prefill-token-threshold: 1024 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb200-2p1d-dep16-decode-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 16 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 256 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - max-cudagraph-capture-size: 256 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - # Avoid concurrent readers observing a mismatched mmap data/index pair. - AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep8-mtp.yaml deleted file mode 100644 index fb540cdbf5..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep8-mtp.yaml +++ /dev/null @@ -1,130 +0,0 @@ -schema: 2 -name: dsv4-gb200-vllm-agentic-mtp-disagg-2p1d-dep8-dep8 - -model: - path: deepseek-v4-pro - container: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 - precision: fp4 - -identity: - model: {repo: deepseek-ai/DeepSeek-V4-Pro} - container: {image: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106} - frameworks: {dynamo: "1.3.1"} - -dynamo: {install: true, source: {pypi: "1.3.1"}} -setup_script: vllm-container-deps.sh -environment: {ETCD_LEASE_TTL: "7200"} -slurm: {time_limit: "8:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} - -resources: - gpu_type: gb200 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - -frontend: - type: dynamo - enable_multiple_frontends: false - env: - # Long AgentX prefills can exceed Dynamo's request-plane default while - # the healthy DEP8 workers are still computing their first responses. - DYN_TCP_REQUEST_TIMEOUT: "60" - args: - router-mode: kv - router-reset-states: true - router-session-affinity-ttl-secs: 14400 - kv-cache-block-size: 256 - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 4 - workers: 2 - gpus: 8 - env: &worker_environment - HF_HUB_CACHE: /hf_hub_cache - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_USE_NCCL_SYMM_MEM: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - args: &dep8_config - kv-transfer-config: '{"kv_connector":"NixlConnector","kv_role":"kv_both"}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:20080","enable_kv_cache_events":true}' - served-model-name: deepseek-ai/DeepSeek-V4-Pro - kv-cache-dtype: fp8 - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - enable-expert-parallel: true - enable-ep-weight-filter: true - moe-backend: deep_gemm_mega_moe - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - max-model-len: 1048576 - max-num-seqs: 16 - max-num-batched-tokens: 16384 - trust-remote-code: true - enforce-eager: true - block-size: 256 - gpu-memory-utilization: 0.95 - no-disable-hybrid-kv-cache-manager: true - kv-cache-metrics: true - enable-sleep-mode: true - tokenizer-mode: deepseek_v4 - speculative-config: '{"method":"mtp","num_speculative_tokens":2}' - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: *worker_environment - args: - <<: *dep8_config - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_fp4_indexer_cache":true}' - enforce-eager: false - gpu-memory-utilization: 0.90 - max-num-seqs: 1024 - max-num-batched-tokens: 1024 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - max-cudagraph-capture-size: 1024 - stream-interval: 10 - -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..b633a32573 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml @@ -0,0 +1,1437 @@ +# srt-slurm recipes for dsv4/vllm/gb200-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: deepseek-v4-pro + precision: fp4 + identity: + model: + repo: deepseek-ai/DeepSeek-V4-Pro + container: {} + frameworks: {} + dynamo: + install: true + source: {} + setup_script: vllm-container-deps.sh + environment: {} + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 2160 + interval_seconds: 10 + resources: + gpu_type: gb200 + gpus_per_node: 4 + frontend: + type: dynamo + enable_multiple_frontends: false + args: {} + engine: + type: vllm + connector: null + roles: {} + sbatch_directives: {} + srun_options: + container-remap-root: '' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + +override_agg_dep8_mtp: + name: dsv4-gb200-vllm-agentic-mtp-agg-dep8 + model: + container: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 + identity: + container: + image: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 + frameworks: + dynamo: 1.3.1 + dynamo: + source: + pypi: 1.3.1 + environment: + ETCD_LEASE_TTL: '7200' + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + args: + router-mode: kv + router-reset-states: true + router-session-affinity-ttl-secs: 14400 + kv-cache-block-size: 256 + engine: + dp_launch_mode: per_node + roles: + agg: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_NCCL_SYMM_MEM: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + args: + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:20080","enable_kv_cache_events":true}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + enable-expert-parallel: true + enable-ep-weight-filter: true + moe-backend: deep_gemm_mega_moe + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + max-model-len: 1048576 + max-num-seqs: 256 + max-num-batched-tokens: 8192 + trust-remote-code: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","custom_ops":["all"]}' + speculative-config: '{"method":"mtp","num_speculative_tokens":2}' + gpu-memory-utilization: 0.9 + stream-interval: 10 + no-disable-hybrid-kv-cache-manager: true + kv-cache-metrics: true + tokenizer-mode: deepseek_v4 + sbatch_directives: + cpus-per-task: '144' + mem: '0' + benchmark: + env: + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' + +# (benchmark.env.IS_MULTINODE) Keep aggregate workers in the multinode result schema so ingestion uses +# (benchmark.env.IS_MULTINODE) the zero decode-worker count instead of duplicating TP into P and D. +override_agg_tp8_c4_mtp3: + name: svf-vllm-agg-gb200-tp8-c4-mtp2-agentic + # GB200 AgentX aggregate topology: one TP8 worker spans two four-GPU + # nodes and serves both prefill and decode at concurrency 4. Keep at least + # 16 sequence slots and otherwise size the scheduler at 4x concurrency. + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.2.1 + dynamo: + source: + wheel: 1.2.1 + environment: + # The frontend shares Grace CPU capacity with the long TP8 cold start. + ETCD_LEASE_TTL: '7200' + DYNAMO_WHEEL_DIRS: /srtctl-wheels + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 150GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: kv + router-reset-states: true + router-temperature: 0.0 + router-queue-threshold: 65536 + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + tokenizer: fastokens + roles: + agg: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '16' + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '0' + TORCH_SYMMMEM: NVSHMEM + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_FLASHINFER_ALLREDUCE_BACKEND: auto + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_SPARSE_INDEXER_MAX_LOGITS_MB: '1024' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + NCCL_P2P_LEVEL: NVL + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb200-agg-mtp2-{job_id} + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 8 + pipeline-parallel-size: 1 + disable-custom-all-reduce: true + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + max-model-len: 1048576 + max-num-seqs: 16 + max-num-batched-tokens: 8192 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64],"mode":0}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + gpu-memory-utilization: 0.94 + stream-interval: 10 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + sbatch_directives: + cpus-per-task: '144' + mem: '0' + +override_agg_tp8_c8_mtp3: + name: svf-vllm-agg-gb200-tp8-c8-mtp2-agentic + # GB200 AgentX aggregate topology: one TP8 worker spans two four-GPU + # nodes and serves both prefill and decode at concurrency 8. Size max-num-seqs at + # 4x concurrency and expand the MTP CUDA-graph envelope to match. + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.2.1 + dynamo: + source: + wheel: 1.2.1 + environment: + ETCD_LEASE_TTL: '7200' + DYNAMO_WHEEL_DIRS: /srtctl-wheels + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 150GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: kv + router-reset-states: true + router-temperature: 0.0 + router-queue-threshold: 65536 + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + tokenizer: fastokens + roles: + agg: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '32' + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '0' + TORCH_SYMMMEM: NVSHMEM + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_FLASHINFER_ALLREDUCE_BACKEND: auto + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_SPARSE_INDEXER_MAX_LOGITS_MB: '1024' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + NCCL_P2P_LEVEL: NVL + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb200-agg-tp8-c8-mtp2-{job_id} + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 8 + pipeline-parallel-size: 1 + disable-custom-all-reduce: true + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + max-model-len: 1048576 + max-num-seqs: 32 + max-num-batched-tokens: 8192 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128],"mode":0}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + gpu-memory-utilization: 0.94 + stream-interval: 10 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + sbatch_directives: + cpus-per-task: '144' + mem: '0' + benchmark: + env: + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: 'false' + +override_agg_tp8_mtp: + name: dsv4-gb200-vllm-agentic-mtp-agg-tp8 + model: + container: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 + identity: + container: + image: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 + frameworks: + dynamo: 1.3.1 + dynamo: + source: + pypi: 1.3.1 + environment: + ETCD_LEASE_TTL: '7200' + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + args: + router-mode: kv + router-reset-states: true + router-session-affinity-ttl-secs: 14400 + kv-cache-block-size: 256 + roles: + agg: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_NCCL_SYMM_MEM: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + args: + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:20080","enable_kv_cache_events":true}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 8 + pipeline-parallel-size: 1 + disable-custom-all-reduce: true + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + max-model-len: 1048576 + max-num-seqs: 128 + max-num-batched-tokens: 8192 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + speculative-config: '{"method":"mtp","num_speculative_tokens":2}' + gpu-memory-utilization: 0.9 + stream-interval: 10 + no-disable-hybrid-kv-cache-manager: true + kv-cache-metrics: true + tokenizer-mode: deepseek_v4 + sbatch_directives: + cpus-per-task: '144' + mem: '0' + benchmark: + env: + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' + +override_disagg_1p1d_dep8_dep8_c128_mtp3: + name: svf-vllm-disagg-gb200-1p1d-dep8-dep8-c128-mtp2-agentic + # GB200 AgentX MTP3 topology: one DEP8 prefill worker feeds one + # DEP8 decode worker at concurrency 128. + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.3.0.dev20260720 + dynamo: + source: + wheel: 1.3.0.dev20260720 + environment: + # Mooncake prefix-block hashes must match across processes and nodes. + PYTHONHASHSEED: '0' + resources: + het_jobs: false + spread_workers: false + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 140GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: random + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + DYN_TCP_CHANNEL_BUFFER: '128' + DYN_TCP_REQUEST_TIMEOUT: '60' + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb200-1p1d-dep8-dep8-c128-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 8192 + long-prefill-token-threshold: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + max-cudagraph-capture-size: 256 + gpu-memory-utilization: 0.85 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + decode: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb200-1p1d-dep8-dep8-c128-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 128 + max-num-batched-tokens: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 512 + gpu-memory-utilization: 0.92 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + sbatch_directives: + cpus-per-task: '72' + benchmark: + env: + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: 'false' + +override_disagg_1p1d_dep8_dep8_c256_mtp3: + name: svf-vllm-disagg-gb200-1p1d-dep8-dep8-c256-mtp2-agentic + # GB200 AgentX MTP3 topology: one DEP8 prefill worker feeds one + # DEP8 decode worker at concurrency 256. + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.3.0.dev20260720 + dynamo: + source: + wheel: 1.3.0.dev20260720 + environment: + # Mooncake prefix-block hashes must match across processes and nodes. + PYTHONHASHSEED: '0' + resources: + het_jobs: false + spread_workers: false + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 140GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: random + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + DYN_TCP_CHANNEL_BUFFER: '128' + DYN_TCP_REQUEST_TIMEOUT: '60' + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb200-1p1d-dep8-dep8-c256-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 8192 + long-prefill-token-threshold: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + max-cudagraph-capture-size: 256 + gpu-memory-utilization: 0.85 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + decode: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb200-1p1d-dep8-dep8-c256-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 128 + max-num-batched-tokens: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 512 + gpu-memory-utilization: 0.92 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + sbatch_directives: + cpus-per-task: '72' + benchmark: + env: + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: 'false' + +override_disagg_1p1d_dep8_dep8_mtp: + name: dsv4-gb200-vllm-agentic-mtp-disagg-1p1d-dep8-dep8 + model: + container: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 + identity: + container: + image: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 + frameworks: + dynamo: 1.3.1 + dynamo: + source: + pypi: 1.3.1 + environment: + ETCD_LEASE_TTL: '7200' + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + args: + router-mode: kv + router-reset-states: true + router-session-affinity-ttl-secs: 14400 + kv-cache-block-size: 256 + env: + # Long AgentX prefills can exceed Dynamo's request-plane default while + # the healthy DEP8 worker is still computing the first response. + DYN_TCP_REQUEST_TIMEOUT: '60' + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_NCCL_SYMM_MEM: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + args: + kv-transfer-config: '{"kv_connector":"NixlConnector","kv_role":"kv_both"}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:20080","enable_kv_cache_events":true}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + enable-expert-parallel: true + enable-ep-weight-filter: true + moe-backend: deep_gemm_mega_moe + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + max-model-len: 1048576 + max-num-seqs: 16 + max-num-batched-tokens: 16384 + trust-remote-code: true + enforce-eager: true + block-size: 256 + gpu-memory-utilization: 0.95 + no-disable-hybrid-kv-cache-manager: true + kv-cache-metrics: true + enable-sleep-mode: true + tokenizer-mode: deepseek_v4 + speculative-config: '{"method":"mtp","num_speculative_tokens":2}' + decode: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_NCCL_SYMM_MEM: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + args: + kv-transfer-config: '{"kv_connector":"NixlConnector","kv_role":"kv_both"}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:20080","enable_kv_cache_events":true}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + enable-expert-parallel: true + enable-ep-weight-filter: true + moe-backend: deep_gemm_mega_moe + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_fp4_indexer_cache":true}' + max-model-len: 1048576 + max-num-seqs: 1024 + max-num-batched-tokens: 1024 + trust-remote-code: true + enforce-eager: false + block-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + kv-cache-metrics: true + enable-sleep-mode: true + tokenizer-mode: deepseek_v4 + speculative-config: '{"method":"mtp","num_speculative_tokens":2}' + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 1024 + stream-interval: 10 + sbatch_directives: + cpus-per-task: '144' + mem: '0' + benchmark: + env: + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' + +override_disagg_2p1d_dep8_dep12_c576_mtp3: + name: svf-vllm-disagg-gb200-2p1d-dep8-dep12-c576-mtp2-agentic + # GB200 AgentX MTP3 topology: two DEP8 prefill workers feed one + # DEP12 decode worker at concurrency 576. Decode consumes P/D KV through NIXL + # and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.3.0.dev20260720 + dynamo: + source: + wheel: 1.3.0.dev20260720 + environment: + # Mooncake prefix-block hashes must match across processes and nodes. + PYTHONHASHSEED: '0' + resources: + het_jobs: false + spread_workers: false + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 140GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: random + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + DYN_TCP_CHANNEL_BUFFER: '128' + DYN_TCP_REQUEST_TIMEOUT: '60' + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 4 + workers: 2 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_MOONCAKE_STORE_SEND_THREADS: '8' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb200-2p1d-prefill-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 8192 + long-prefill-token-threshold: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + decode: + nodes: 3 + workers: 1 + gpus: 12 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb200-2p1d-dep12-decode-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 12 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 256 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + sbatch_directives: + cpus-per-task: '72' + mem: '0' + benchmark: + env: + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: 'false' + +override_disagg_2p1d_dep8_dep16_c512_mtp3: + name: svf-vllm-disagg-gb200-2p1d-dep8-dep16-c512-mtp2-agentic + # GB200 AgentX MTP3 topology: two DEP8 prefill workers feed one + # and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. + # DEP16 decode worker at concurrency 512. Decode consumes P/D KV through NIXL + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.3.0.dev20260720 + dynamo: + source: + wheel: 1.3.0.dev20260720 + environment: + # Mooncake prefix-block hashes must match across processes and nodes. + PYTHONHASHSEED: '0' + resources: + het_jobs: false + spread_workers: false + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 140GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: random + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + DYN_TCP_CHANNEL_BUFFER: '128' + DYN_TCP_REQUEST_TIMEOUT: '60' + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 4 + workers: 2 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_MOONCAKE_STORE_SEND_THREADS: '8' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb200-2p1d-prefill-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 8192 + long-prefill-token-threshold: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + decode: + nodes: 4 + workers: 1 + gpus: 16 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb200-2p1d-dep16-decode-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 16 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 256 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + sbatch_directives: + cpus-per-task: '72' + mem: '0' + benchmark: + env: + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: 'false' + +override_disagg_2p1d_dep8_dep8_mtp: + name: dsv4-gb200-vllm-agentic-mtp-disagg-2p1d-dep8-dep8 + model: + container: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 + identity: + container: + image: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 + frameworks: + dynamo: 1.3.1 + dynamo: + source: + pypi: 1.3.1 + environment: + ETCD_LEASE_TTL: '7200' + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + args: + router-mode: kv + router-reset-states: true + router-session-affinity-ttl-secs: 14400 + kv-cache-block-size: 256 + env: + # Long AgentX prefills can exceed Dynamo's request-plane default while + # the healthy DEP8 workers are still computing their first responses. + DYN_TCP_REQUEST_TIMEOUT: '60' + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 4 + workers: 2 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_NCCL_SYMM_MEM: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + args: + kv-transfer-config: '{"kv_connector":"NixlConnector","kv_role":"kv_both"}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:20080","enable_kv_cache_events":true}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + enable-expert-parallel: true + enable-ep-weight-filter: true + moe-backend: deep_gemm_mega_moe + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + max-model-len: 1048576 + max-num-seqs: 16 + max-num-batched-tokens: 16384 + trust-remote-code: true + enforce-eager: true + block-size: 256 + gpu-memory-utilization: 0.95 + no-disable-hybrid-kv-cache-manager: true + kv-cache-metrics: true + enable-sleep-mode: true + tokenizer-mode: deepseek_v4 + speculative-config: '{"method":"mtp","num_speculative_tokens":2}' + decode: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_NCCL_SYMM_MEM: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + args: + kv-transfer-config: '{"kv_connector":"NixlConnector","kv_role":"kv_both"}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:20080","enable_kv_cache_events":true}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + enable-expert-parallel: true + enable-ep-weight-filter: true + moe-backend: deep_gemm_mega_moe + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_fp4_indexer_cache":true}' + max-model-len: 1048576 + max-num-seqs: 1024 + max-num-batched-tokens: 1024 + trust-remote-code: true + enforce-eager: false + block-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + kv-cache-metrics: true + enable-sleep-mode: true + tokenizer-mode: deepseek_v4 + speculative-config: '{"method":"mtp","num_speculative_tokens":2}' + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 1024 + stream-interval: 10 + sbatch_directives: + cpus-per-task: '144' + mem: '0' + benchmark: + env: + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/agg-tp4-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/agg-tp4-mtp.yaml deleted file mode 100644 index dda2443295..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/agg-tp4-mtp.yaml +++ /dev/null @@ -1,159 +0,0 @@ -schema: 2 -name: "svf-vllm-agg-gb300-tp4-mtp-agentic" - -# GB300 AgentX aggregate topology: one TP4 worker occupies one four-GPU node -# and serves both prefill and decode at concurrency 8. Size max-num-seqs at -# 4x concurrency and expand the MTP CUDA-graph envelope to match. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.2.1" - -dynamo: - install: true - - source: - wheel: "1.2.1" -environment: - DYNAMO_WHEEL_DIRS: "/srtctl-wheels" - ETCD_LEASE_TTL: "7200" - -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "150GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "kv" - router-reset-states: true - router-temperature: 0.0 - router-queue-threshold: 65536 - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - tokenizer: "fastokens" - -engine: - type: vllm - connector: -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "32" - TILELANG_CLEANUP_TEMP_FILES: "1" - VLLM_USE_NCCL_SYMM_MEM: "0" - TORCH_SYMMMEM: "NVSHMEM" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_USE_RUST_FRONTEND: "1" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "auto" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_SPARSE_INDEXER_MAX_LOGITS_MB: "1024" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - NCCL_P2P_LEVEL: "NVL" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-agg-tp4-mtp-{job_id}" - args: - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 4 - pipeline-parallel-size: 1 - disable-custom-all-reduce: true - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - max-model-len: 1048576 - max-num-seqs: 32 - max-num-batched-tokens: 8192 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128],"mode":0}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - gpu-memory-utilization: 0.94 - stream-interval: 10 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - # Avoid concurrent readers observing a mismatched mmap data/index pair. - AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/agg-tp8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/agg-tp8-mtp.yaml deleted file mode 100644 index f17faead2a..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/agg-tp8-mtp.yaml +++ /dev/null @@ -1,160 +0,0 @@ -schema: 2 -name: "svf-vllm-agg-gb300-tp8-mtp-agentic" - -# GB300 AgentX aggregate topology: one TP8 worker spans two four-GPU -# nodes and serves both prefill and decode at concurrency 4. Keep at least -# 16 sequence slots and otherwise size the scheduler at 4x concurrency. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.2.1" - -dynamo: - install: true - - source: - wheel: "1.2.1" -environment: - DYNAMO_WHEEL_DIRS: "/srtctl-wheels" - # The frontend shares Grace CPU capacity with the long TP8 cold start. - ETCD_LEASE_TTL: "7200" - -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "150GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "kv" - router-reset-states: true - router-temperature: 0.0 - router-queue-threshold: 65536 - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - tokenizer: "fastokens" - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "16" - TILELANG_CLEANUP_TEMP_FILES: "1" - VLLM_USE_NCCL_SYMM_MEM: "0" - TORCH_SYMMMEM: "NVSHMEM" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_USE_RUST_FRONTEND: "1" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "auto" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_SPARSE_INDEXER_MAX_LOGITS_MB: "1024" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - NCCL_P2P_LEVEL: "NVL" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-agg-mtp-{job_id}" - args: - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 8 - pipeline-parallel-size: 1 - disable-custom-all-reduce: true - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - max-model-len: 1048576 - max-num-seqs: 16 - max-num-batched-tokens: 8192 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64],"mode":0}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - gpu-memory-utilization: 0.94 - stream-interval: 10 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - -sbatch_directives: - cpus-per-task: "144" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - # Keep aggregate workers in the multinode result schema so ingestion uses - # the zero decode-worker count instead of duplicating TP into P and D. - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep16-c128-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep16-c128-mtp.yaml deleted file mode 100644 index acdca7c858..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep16-c128-mtp.yaml +++ /dev/null @@ -1,228 +0,0 @@ -# Derived from PR #2302's DEP8/DEP32 c256 recipe. -# Halve the prefill/decode topology while preserving its conservative -# 4/16/16 decode limits and isolated JIT cache paths. -schema: 2 -name: "svf-vllm-disagg-gb300-1p1d-dep4-dep16-c128-mtp-agentic" - -# GB300 AgentX MTP3 topology: one DEP4 prefill worker feeds one -# DEP16 decode worker at concurrency 128. Decode consumes P/D KV through NIXL -# and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.4.0" - -dynamo: - install: true - - source: - wheel: "1.4.0" -environment: - # Mooncake prefix-block hashes must match across processes and nodes. - PYTHONHASHSEED: "0" - -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "180GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "random" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TCP_CHANNEL_BUFFER: "128" - DYN_TCP_REQUEST_TIMEOUT: "60" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_MOONCAKE_STORE_SEND_THREADS: "8" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_USE_BREAKABLE_CUDAGRAPH: "0" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep8-prefill-c128-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - safetensors-load-strategy: "prefetch" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 4 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 256 - max-num-batched-tokens: 8192 - long-prefill-token-threshold: 1024 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep32-decode-c128-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - safetensors-load-strategy: "prefetch" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 16 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 4 - max-num-batched-tokens: 16 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - max-cudagraph-capture-size: 16 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - # Avoid concurrent readers observing a mismatched mmap data/index pair. - AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep16-c256-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep16-c256-mtp.yaml deleted file mode 100644 index bb98740ab7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep16-c256-mtp.yaml +++ /dev/null @@ -1,228 +0,0 @@ -# Derived from PR #2302's DEP8/DEP32 c256 recipe. -# Halve the prefill/decode topology while preserving its conservative -# 8/32/32 decode limits and isolated JIT cache paths. -schema: 2 -name: "svf-vllm-disagg-gb300-1p1d-dep4-dep16-c256-mtp-agentic" - -# GB300 AgentX MTP3 topology: one DEP4 prefill worker feeds one -# DEP16 decode worker at concurrency 256. Decode consumes P/D KV through NIXL -# and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.4.0" - -dynamo: - install: true - - source: - wheel: "1.4.0" -environment: - # Mooncake prefix-block hashes must match across processes and nodes. - PYTHONHASHSEED: "0" - -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "180GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "random" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TCP_CHANNEL_BUFFER: "128" - DYN_TCP_REQUEST_TIMEOUT: "60" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_MOONCAKE_STORE_SEND_THREADS: "8" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_USE_BREAKABLE_CUDAGRAPH: "0" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep8-prefill-c256-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - safetensors-load-strategy: "prefetch" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 4 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 256 - max-num-batched-tokens: 8192 - long-prefill-token-threshold: 1024 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep32-decode-c256-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - safetensors-load-strategy: "prefetch" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 16 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 8 - max-num-batched-tokens: 32 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - max-cudagraph-capture-size: 32 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - # Avoid concurrent readers observing a mismatched mmap data/index pair. - AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep8-c256-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep8-c256-mtp.yaml deleted file mode 100644 index 62867c6566..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep8-c256-mtp.yaml +++ /dev/null @@ -1,221 +0,0 @@ -schema: 2 -name: "svf-vllm-disagg-gb300-1p1d-dep4-dep8-c256-mtp-agentic" - -# GB300 AgentX MTP3 topology: one DEP4 prefill worker feeds one -# DEP8 decode worker at concurrency 256. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.3.0.dev20260720" - -dynamo: - install: true - - source: - wheel: "1.3.0.dev20260720" -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "180GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -environment: - # Mooncake prefix-block hashes must match across processes and nodes. - PYTHONHASHSEED: "0" - -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "random" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TCP_CHANNEL_BUFFER: "128" - DYN_TCP_REQUEST_TIMEOUT: "60" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_USE_BREAKABLE_CUDAGRAPH: "0" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep4-dep8-c256-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 4 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 8192 - long-prefill-token-threshold: 1024 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - max-cudagraph-capture-size: 256 - gpu-memory-utilization: 0.92 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep4-dep8-c256-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 256 - max-num-batched-tokens: 1024 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - max-cudagraph-capture-size: 1024 - gpu-memory-utilization: 0.92 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] -sbatch_directives: - cpus-per-task: "72" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - # Avoid concurrent readers observing a mismatched mmap data/index pair. - AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep8-dep16-c512-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep8-dep16-c512-mtp.yaml deleted file mode 100644 index d9d280a6c9..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep8-dep16-c512-mtp.yaml +++ /dev/null @@ -1,228 +0,0 @@ -# Source: https://github.com/SemiAnalysisAI/InferenceX/blob/0c33d4615792705ed12bfc204e3a54cfa436cf02/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb300-1p1d-dep8-dep16-c512-mtp-agentic.yaml -# Runtime and performance arguments follow the current DEP8/DEP16 GB300 P/D -# recipes; throughput-only synthetic MTP acceptance is injected at launch. -schema: 2 -name: "svf-vllm-disagg-gb300-1p1d-dep8-dep16-c512-mtp-agentic" - -# GB300 AgentX MTP3 topology: one DEP8 prefill worker feeds one -# DEP16 decode worker at concurrency 512. Decode consumes P/D KV through NIXL -# and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.4.0" - -dynamo: - install: true - - source: - wheel: "1.4.0" -environment: - # Mooncake prefix-block hashes must match across processes and nodes. - PYTHONHASHSEED: "0" - -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "180GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "random" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TCP_CHANNEL_BUFFER: "128" - DYN_TCP_REQUEST_TIMEOUT: "60" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_MOONCAKE_STORE_SEND_THREADS: "8" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_USE_BREAKABLE_CUDAGRAPH: "0" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep8-prefill-c512-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - safetensors-load-strategy: "prefetch" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 16384 - long-prefill-token-threshold: 1024 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep16-decode-c512-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - safetensors-load-strategy: "prefetch" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 16 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 256 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - max-cudagraph-capture-size: 256 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - # Avoid concurrent readers observing a mismatched mmap data/index pair. - AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p4d-dep4-tp8-c4-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p4d-dep4-tp8-c4-mtp.yaml deleted file mode 100644 index 922a5343a8..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p4d-dep4-tp8-c4-mtp.yaml +++ /dev/null @@ -1,234 +0,0 @@ -# Source: https://github.com/Inferact/srt-slurm-sa/blob/4a870cd5bc333bf8a312fc65ca1ea82cdab9b2df/recipes/vllm/deepseek-v4-pro/GB300/agentic/agentx-v1.0.1/1p3d-pdep4-dtp8-c3-kv-nixl-mtp-flashinfer-ar-lpt512-psi1.yaml -# Adapted from the source topology to a 1P4D concurrency-4 point for the -# InferenceX AgentX harness; eval-only runs continue to verify real MTP output. -schema: 2 -name: "svf-vllm-disagg-gb300-1p4d-dep4-tp8-c4-mtp-agentic" - -# GB300 high-interactivity AgentX MTP3 topology: one DEP4 prefill worker -# feeds four TP8 decode workers at concurrency 4 through NIXL. -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.4.0" - -dynamo: - install: true - source: - wheel: "1.4.0" - request_plane: tcp - -environment: - DYNAMO_WHEEL_DIRS: "/srtctl-wheels" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "auto" - VLLM_USE_NCCL_SYMM_MEM: "0" - TORCH_SYMMMEM: "NVSHMEM" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p4d-{job_id}" - # Multi-node TP8 decode spans two GB300 nodes; match the sibling recipes' - # NCCL/UCX fabric settings (MNNVL/NVLS, IB HCAs). - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - -slurm: - time_limit: "08:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - MODEL: "deepseek-ai/DeepSeek-V4-Pro" - MODEL_PREFIX: "dsv4" - FRAMEWORK: "dynamo-vllm" - PRECISION: "fp4" - CONC: "4" - DURATION: "3600" - KV_OFFLOADING: "none" - ETCD_LEASE_TTL: "120" - DYN_ROUTER_QUEUE_THRESHOLD: "None" - DYN_TOKENIZER_CACHE: "1" - DYN_TOKENIZER_CACHE_BYTES: "8000000000" - DYN_TOKENIZER: "fastokens" - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: "14400" - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - args: - router-mode: "kv" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - tokenizer: "fastokens" - placement: - node: first_decode -engine: - type: vllm - connector: - dp_launch_mode: per_node - # vLLM KV routing needs prefill KV events so Dynamo can select the - # cache-owning PDEP4 rank. -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - VLLM_USE_BREAKABLE_CUDAGRAPH: "0" - - args: - kv-transfer-config: '{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - safetensors-load-strategy: "prefetch" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 4 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 8192 - long-prefill-token-threshold: 512 - prefill-schedule-interval: 1 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - gpu-memory-utilization: 0.92 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - # MTP capture sizes are tokens: 64 seqs * (1 target + 3 drafts). - max-cudagraph-capture-size: 256 - kv_events: true - decode: - nodes: 8 - workers: 4 - gpus: 8 - - args: - kv-transfer-config: '{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}}' - # TP8 spans two GB300 nodes; custom all-reduce is single-node only, so - # decode uses the FlashInfer allreduce path like the agg TP8 recipes. - disable-custom-all-reduce: true - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - safetensors-load-strategy: "prefetch" - kv-cache-dtype: "fp8" - tensor-parallel-size: 8 - pipeline-parallel-size: 1 - max-model-len: 1048576 - max-num-seqs: 16 - max-num-batched-tokens: 64 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - # MTP capture sizes are tokens: 16 seqs * (1 target + 3 drafts). - max-cudagraph-capture-size: 64 - gpu-memory-utilization: 0.92 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' -setup_script: vllm-container-deps.sh - -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - # The AgentX client uses localhost:8000, so colocate it with the Dynamo - # frontend launched on the first decode node. - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - MODEL: "deepseek-ai/DeepSeek-V4-Pro" - MODEL_PREFIX: "dsv4" - SERVED_MODEL_NAME: "deepseek-ai/DeepSeek-V4-Pro" - FRAMEWORK: "dynamo-vllm" - PRECISION: "fp4" - CONC: "4" - DURATION: "3600" - RUNNER_TYPE: "gb300" - IMAGE: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - SPEC_DECODING: "mtp" - DISAGG: "true" - OFFLOADING: "none" - KV_OFFLOADING: "none" - TP: "8" - PREFILL_TP: "1" - PREFILL_NUM_WORKERS: "1" - PREFILL_EP: "4" - DECODE_TP: "8" - DECODE_NUM_WORKERS: "4" - DECODE_EP: "1" - EP_SIZE: "1" - AIPERF_MAX_OSL: "none" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - # Avoid concurrent readers observing a mismatched mmap data/index pair. - AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" - NUM_DATASET_ENTRIES: "393" - HF_WEKA_DATASET: "semianalysisai/cc-traces-weka-062126" - PUBLIC_DATASET: "semianalysis_cc_traces_weka_062126" - placement: - node: first_decode diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p6d-dep4-tp4.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p6d-dep4-tp4.yaml deleted file mode 100644 index 2670488be1..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p6d-dep4-tp4.yaml +++ /dev/null @@ -1,179 +0,0 @@ -schema: 2 -name: "svf-vllm-disagg-gb300-1p6d-dep4-tp4-agentic" - -# Agentic-coding variant of dsv4/vllm/gb300-fp4/8k1k/disagg-1p6d-dep4-tp4-stp.yaml. -# Topology is identical (1 prefill DEP=4 + 6 decode TP=4, 28 GPUs across 7 -# GB300 nodes + 1 dedicated NATS/etcd infra node) so we can compare against -# the fixed-seq-len 1p6d baseline at the same concurrency point (192). -# -# Divergence vs the 8k1k sibling: -# - benchmark.type: sa-bench -> custom (hands off to agentic_srt.sh) -# - max-model-len: removed (let vLLM derive from model config; agentic -# trajectories blow past any small explicit cap) -# - no-enable-prefix-caching: dropped (prefix caching MUST be on for -# trajectory reuse — entire point of agentic) -# Note: --enable-auto-tool-choice / --tool-call-parser / --reasoning-parser -# are NOT set on the worker. The dynamo-vllm worker entrypoint doesn't -# accept them (different arg parser than `vllm serve`). In disagg, chat -# parsing happens at the dynamo frontend, not at the worker. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:v0.21.0-ubuntu2404" - precision: "fp4" - -dynamo: - install: true - source: - wheel: "1.2.0.dev20260426" - -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - TILELANG_CLEANUP_TEMP_FILES: "1" - VLLM_USE_NCCL_SYMM_MEM: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - TORCH_SYMMMEM: "NVSHMEM" - args: - kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 4 - data-parallel-rpc-port: 13345 - enable-expert-parallel: true - attention-config: '{"use_fp4_indexer_cache": true}' - moe-backend: "deep_gemm_mega_moe" - # enforce-eager: true - # max-num-seqs: 256 - max-num-batched-tokens: 16384 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - gpu-memory-utilization: 0.9 - enable-ep-weight-filter: true - no-disable-hybrid-kv-cache-manager: true - enable-sleep-mode: true - tokenizer-mode: deepseek_v4 - decode: - nodes: 6 - workers: 6 - gpus: 4 - - env: - TILELANG_CLEANUP_TEMP_FILES: "1" - VLLM_USE_NCCL_SYMM_MEM: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - TORCH_SYMMMEM: "NVSHMEM" - - args: - kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 4 - pipeline-parallel-size: 1 - # max-num-seqs: 512 - trust-remote-code: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - gpu-memory-utilization: 0.9 - stream-interval: 10 - no-disable-hybrid-kv-cache-manager: true - enable-ep-weight-filter: true - all2all-backend: "flashinfer_nvlink_one_sided" - no-enable-flashinfer-autotune: true - enable-sleep-mode: true - tokenizer-mode: deepseek_v4 - -# sbatch + srun resource grants for clusters without per-GPU defaults. -# -# mem=0: allocate all available node memory (~868 GB on CW gb300). Without -# this, sbatch only requests ntasks × DefMemPerCPU = 8 × 4 GB = 32 GB for -# the whole job and worker cgroups OOM-kill mid model load (R7-R11 hit -# this; sacct showed AllocTRES mem=4G per step). -# -# cpus-per-task=72: give each task one CW gb300 NUMA socket (144 cores -# split 2 × 72). Critical for the *infra step* (etcd + nats) which -# srtctl spawns without --gres=gpu — on CW that means DefMemPerCPU -# applies and the step gets 1 CPU by default. With 24 dynamo DP ranks -# all hammering etcd for lease keep-alives, single-CPU etcd can't keep -# up and dies (R12 hit this; etcd reported max-cpu-set=1, leases -# deadline-exceeded, infra SIGKILL'd at 16:35:49). 72 CPUs is plenty -# for both etcd + nats AND for vLLM worker auxiliary threads. -# -# nv gb300 doesn't need this because cluster default DefCpuPerGPU=35 -# auto-allocates 4*35=140 CPUs per GPU-bearing task; cw has no per-GPU -# default. Setting it here is safe on both because the value is ≤ node -# CPU count. -# -# srun_options.mem=0 forces each srun step to use the full node memory -# (without it, srun steps default back to cpus_per_task × DefMemPerCPU). -# Docs: docs/config-reference.md#sbatch_directives + #srun_options. -sbatch_directives: - mem: "0" - cpus-per-task: "72" -srun_options: - mem: "0" - # gb300-nv: pyxis maps the calling user (sa-shared) into the container as - # uid 345200007. dpkg refuses to run without EUID 0 even though - # ENROOT_ROOTFS_WRITABLE=1 makes the rootfs writable, so the agentic_srt - # apt-get install git step fails. --container-remap-root asks pyxis to - # remap us to uid 0 inside the container. srt-slurm renders empty-string - # values as flag-only srun args (see core/slurm.py:250). - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - # Container-side path of the aiperf mmap dataset cache; the host-side - # mount is wired via launch_gb300-*.sh's srtslurm.yaml default_mounts. - # Without this, aiperf re-tokenizes + re-writes ~65 GB of mmap files - # per dataset on every run. - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - # Persistent HF hub cache (also wired via default_mounts) so the trace - # dataset isn't re-downloaded on every run. Overrides the workflow-level - # HF_HUB_CACHE=/mnt/hf_hub_cache, which doesn't exist on these nodes. - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep12-c1152-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep12-c1152-mtp.yaml deleted file mode 100644 index 2b234884ff..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep12-c1152-mtp.yaml +++ /dev/null @@ -1,223 +0,0 @@ -schema: 2 -name: "svf-vllm-disagg-gb300-2p1d-dep8-dep12-c1152-mtp-agentic" - -# GB300 AgentX MTP3 topology: two DEP8 prefill workers feed one -# DEP12 decode worker at concurrency 1152. Decode consumes P/D KV through NIXL -# and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.3.0.dev20260720" - -dynamo: - install: true - - source: - wheel: "1.3.0.dev20260720" -environment: - # Mooncake prefix-block hashes must match across processes and nodes. - PYTHONHASHSEED: "0" - -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "180GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "random" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TCP_CHANNEL_BUFFER: "128" - DYN_TCP_REQUEST_TIMEOUT: "60" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 4 - workers: 2 - gpus: 8 - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_MOONCAKE_STORE_SEND_THREADS: "8" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_USE_BREAKABLE_CUDAGRAPH: "0" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-2p1d-prefill-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 8192 - long-prefill-token-threshold: 1024 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] - decode: - nodes: 3 - workers: 1 - gpus: 12 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-2p1d-dep12-decode-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 12 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 256 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - max-cudagraph-capture-size: 256 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - # Avoid concurrent readers observing a mismatched mmap data/index pair. - AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep16-c1024-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep16-c1024-mtp.yaml deleted file mode 100644 index 7b8b762e6f..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep16-c1024-mtp.yaml +++ /dev/null @@ -1,223 +0,0 @@ -schema: 2 -name: "svf-vllm-disagg-gb300-2p1d-dep8-dep16-c1024-mtp-agentic" - -# GB300 AgentX MTP3 topology: two DEP8 prefill workers feed one -# DEP16 decode worker at concurrency 1024. Decode consumes P/D KV through NIXL -# and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f" - frameworks: - dynamo: "1.3.0.dev20260720" - -dynamo: - install: true - - source: - wheel: "1.3.0.dev20260720" -environment: - # Mooncake prefix-block hashes must match across processes and nodes. - PYTHONHASHSEED: "0" - -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "180GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "random" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TCP_CHANNEL_BUFFER: "128" - DYN_TCP_REQUEST_TIMEOUT: "60" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 4 - workers: 2 - gpus: 8 - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_MOONCAKE_STORE_SEND_THREADS: "8" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_USE_BREAKABLE_CUDAGRAPH: "0" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-2p1d-prefill-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 8192 - long-prefill-token-threshold: 1024 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" - VLLM_DSV4_MEGA_FP8_COMBINE: "1" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-2p1d-dep16-decode-{job_id}" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 16 - data-parallel-rpc-port: 13345 - enable-cumem-allocator: true - enable-expert-parallel: true - enable-ep-weight-filter: true - max-model-len: 1048576 - max-num-seqs: 64 - max-num-batched-tokens: 256 - trust-remote-code: true - no-enable-flashinfer-autotune: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - max-cudagraph-capture-size: 256 - gpu-memory-utilization: 0.90 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: "deepseek_v4" - attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' - moe-backend: "deep_gemm_amxf4_mega_moe" - numa-bind: true - numa-bind-nodes: [0, 0, 1, 1] -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - # Avoid concurrent readers observing a mismatched mmap data/index pair. - AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-4p1d-dep4-dep8-24-c4096.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-4p1d-dep4-dep8-24-c4096.yaml deleted file mode 100644 index 20e836b93b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/disagg-4p1d-dep4-dep8-24-c4096.yaml +++ /dev/null @@ -1,188 +0,0 @@ -schema: 2 -name: "svf-vllm-disagg-gb300-4p1d-dep4-dep8-24-c4096-agentic" - -# Agentic-coding variant of dsv4/vllm/gb300-fp4/8k1k/disagg-4p1d-dep4-dep8-24-c4096-stp.yaml. -# Max-throughput shape: 4 prefill (DEP=4 each) + 1 decode (DEP=8). 6 GB300 -# nodes (4P + 2D = 24 GPUs at 4 GPUs/node) plus a dedicated NATS/etcd infra -# node. Sized for concurrency 4096 with deep_gemm_mega_moe on both workers. -# -# Divergence vs the 8k1k sibling: -# - benchmark.type: sa-bench -> custom (hands off to agentic_srt.sh) -# - max-model-len: removed (let vLLM derive from model config; agentic -# trajectories blow past any small explicit cap) -# - no-enable-prefix-caching: dropped (prefix caching MUST be on for -# trajectory reuse — entire point of agentic) -# Note: --enable-auto-tool-choice / --tool-call-parser / --reasoning-parser -# are NOT set on the worker. The dynamo-vllm worker entrypoint doesn't -# accept them (different arg parser than `vllm serve`). In disagg, chat -# parsing happens at the dynamo frontend, not at the worker. - -model: - path: "deepseek-v4-pro" - container: "vllm/vllm-openai:v0.21.0-ubuntu2404" - precision: "fp4" - -dynamo: - install: true - source: - wheel: "1.2.0.dev20260426" - -setup_script: vllm-container-deps.sh - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - -engine: - type: vllm - connector: - -roles: - prefill: - nodes: 4 - workers: 4 - gpus: 4 - env: - TILELANG_CLEANUP_TEMP_FILES: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_USE_NCCL_SYMM_MEM: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - TORCH_SYMMMEM: "NVSHMEM" - - args: - kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 4 - data-parallel-rpc-port: 13345 - enable-expert-parallel: true - # enforce-eager: true - # Inherited from fixed-sequence recipes; let vLLM select the scheduler - # sequence limit until this is tuned explicitly for the agentic trace. - # max-num-seqs: 16 - max-num-batched-tokens: 16384 - trust-remote-code: true - no-enable-flashinfer-autotune: true - safetensors-load-strategy: "prefetch" - block-size: 256 - gpu-memory-utilization: 0.9 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: deepseek_v4 - enable-ep-weight-filter: true - enable-sleep-mode: true - moe-backend: "deep_gemm_mega_moe" - - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: - TILELANG_CLEANUP_TEMP_FILES: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_USE_NCCL_SYMM_MEM: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - TORCH_SYMMMEM: "NVSHMEM" - - args: - kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}' - served-model-name: "deepseek-ai/DeepSeek-V4-Pro" - kv-cache-dtype: "fp8" - tensor-parallel-size: 1 - pipeline-parallel-size: 1 - data-parallel-size: 8 - data-parallel-rpc-port: 13345 - enable-expert-parallel: true - # max-num-seqs: 512 - trust-remote-code: true - block-size: 256 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' - gpu-memory-utilization: 0.9 - stream-interval: 10 - no-disable-hybrid-kv-cache-manager: true - tokenizer-mode: deepseek_v4 - enable-ep-weight-filter: true - enable-sleep-mode: true - moe-backend: "deep_gemm_mega_moe" - -# sbatch + srun resource grants for clusters without per-GPU defaults. -# -# mem=0: allocate all available node memory (~868 GB on CW gb300). Without -# this, sbatch only requests ntasks × DefMemPerCPU = 8 × 4 GB = 32 GB for -# the whole job and worker cgroups OOM-kill mid model load (R7-R11 hit -# this; sacct showed AllocTRES mem=4G per step). -# -# cpus-per-task=72: give each task one CW gb300 NUMA socket (144 cores -# split 2 × 72). Critical for the *infra step* (etcd + nats) which -# srtctl spawns without --gres=gpu — on CW that means DefMemPerCPU -# applies and the step gets 1 CPU by default. With 24 dynamo DP ranks -# all hammering etcd for lease keep-alives, single-CPU etcd can't keep -# up and dies (R12 hit this; etcd reported max-cpu-set=1, leases -# deadline-exceeded, infra SIGKILL'd at 16:35:49). 72 CPUs is plenty -# for both etcd + nats AND for vLLM worker auxiliary threads. -# -# nv gb300 doesn't need this because cluster default DefCpuPerGPU=35 -# auto-allocates 4*35=140 CPUs per GPU-bearing task; cw has no per-GPU -# default. Setting it here is safe on both because the value is ≤ node -# CPU count. -# -# srun_options.mem=0 forces each srun step to use the full node memory -# (without it, srun steps default back to cpus_per_task × DefMemPerCPU). -# Docs: docs/config-reference.md#sbatch_directives + #srun_options. -sbatch_directives: - mem: "0" - cpus-per-task: "72" -srun_options: - mem: "0" - # gb300-nv: pyxis maps the calling user (sa-shared) into the container as - # uid 345200007. dpkg refuses to run without EUID 0 even though - # ENROOT_ROOTFS_WRITABLE=1 makes the rootfs writable, so the agentic_srt - # apt-get install git step fails. --container-remap-root asks pyxis to - # remap us to uid 0 inside the container. srt-slurm renders empty-string - # values as flag-only srun args (see core/slurm.py:250). - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - # Container-side path of the aiperf mmap dataset cache; the host-side - # mount is wired via launch_gb300-*.sh's srtslurm.yaml default_mounts. - # Without this, aiperf re-tokenizes + re-writes ~65 GB of mmap files - # per dataset on every run. - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - # Persistent HF hub cache (also wired via default_mounts) so the trace - # dataset isn't re-downloaded on every run. Overrides the workflow-level - # HF_HUB_CACHE=/mnt/hf_hub_cache, which doesn't exist on these nodes. - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..bbf83996a5 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml @@ -0,0 +1,1922 @@ +# srt-slurm recipes for dsv4/vllm/gb300-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: deepseek-v4-pro + precision: fp4 + dynamo: + install: true + source: {} + setup_script: vllm-container-deps.sh + slurm: {} + health_check: + interval_seconds: 10 + resources: + gpu_type: gb300 + gpus_per_node: 4 + frontend: + type: dynamo + enable_multiple_frontends: false + engine: + type: vllm + connector: null + roles: {} + sbatch_directives: {} + srun_options: + container-remap-root: '' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + +override_agg_tp4_mtp: + name: svf-vllm-agg-gb300-tp4-mtp-agentic + # GB300 AgentX aggregate topology: one TP4 worker occupies one four-GPU node + # and serves both prefill and decode at concurrency 8. Size max-num-seqs at + # 4x concurrency and expand the MTP CUDA-graph envelope to match. + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + model: + repo: deepseek-ai/DeepSeek-V4-Pro + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.2.1 + dynamo: + source: + wheel: 1.2.1 + environment: + DYNAMO_WHEEL_DIRS: /srtctl-wheels + ETCD_LEASE_TTL: '7200' + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 2160 + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 150GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: kv + router-reset-states: true + router-temperature: 0.0 + router-queue-threshold: 65536 + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + tokenizer: fastokens + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '32' + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '0' + TORCH_SYMMMEM: NVSHMEM + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_FLASHINFER_ALLREDUCE_BACKEND: auto + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_SPARSE_INDEXER_MAX_LOGITS_MB: '1024' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + NCCL_P2P_LEVEL: NVL + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-agg-tp4-mtp-{job_id} + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 4 + pipeline-parallel-size: 1 + disable-custom-all-reduce: true + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + max-model-len: 1048576 + max-num-seqs: 32 + max-num-batched-tokens: 8192 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128],"mode":0}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + gpu-memory-utilization: 0.94 + stream-interval: 10 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + sbatch_directives: + cpus-per-task: '72' + mem: '0' + benchmark: + env: + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: 'false' + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + +override_agg_tp8_mtp: + name: svf-vllm-agg-gb300-tp8-mtp-agentic + # GB300 AgentX aggregate topology: one TP8 worker spans two four-GPU + # nodes and serves both prefill and decode at concurrency 4. Keep at least + # 16 sequence slots and otherwise size the scheduler at 4x concurrency. + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + model: + repo: deepseek-ai/DeepSeek-V4-Pro + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.2.1 + dynamo: + source: + wheel: 1.2.1 + environment: + DYNAMO_WHEEL_DIRS: /srtctl-wheels + # The frontend shares Grace CPU capacity with the long TP8 cold start. + ETCD_LEASE_TTL: '7200' + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 2160 + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 150GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: kv + router-reset-states: true + router-temperature: 0.0 + router-queue-threshold: 65536 + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + tokenizer: fastokens + roles: + agg: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '16' + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '0' + TORCH_SYMMMEM: NVSHMEM + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_FLASHINFER_ALLREDUCE_BACKEND: auto + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_SPARSE_INDEXER_MAX_LOGITS_MB: '1024' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + NCCL_P2P_LEVEL: NVL + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-agg-mtp-{job_id} + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 8 + pipeline-parallel-size: 1 + disable-custom-all-reduce: true + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + max-model-len: 1048576 + max-num-seqs: 16 + max-num-batched-tokens: 8192 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64],"mode":0}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + gpu-memory-utilization: 0.94 + stream-interval: 10 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + sbatch_directives: + cpus-per-task: '144' + mem: '0' + benchmark: + # Keep aggregate workers in the multinode result schema so ingestion uses + # the zero decode-worker count instead of duplicating TP into P and D. + env: + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + +# Derived from PR #2302's DEP8/DEP32 c256 recipe. +# Halve the prefill/decode topology while preserving its conservative +# 4/16/16 decode limits and isolated JIT cache paths. +override_disagg_1p1d_dep4_dep16_c128_mtp: + name: svf-vllm-disagg-gb300-1p1d-dep4-dep16-c128-mtp-agentic + # GB300 AgentX MTP3 topology: one DEP4 prefill worker feeds one + # DEP16 decode worker at concurrency 128. Decode consumes P/D KV through NIXL + # and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + model: + repo: deepseek-ai/DeepSeek-V4-Pro + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.4.0 + dynamo: + source: + wheel: 1.4.0 + environment: + # Mooncake prefix-block hashes must match across processes and nodes. + PYTHONHASHSEED: '0' + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 2160 + resources: + het_jobs: false + spread_workers: false + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 180GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: random + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + DYN_TCP_CHANNEL_BUFFER: '128' + DYN_TCP_REQUEST_TIMEOUT: '60' + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_MOONCAKE_STORE_SEND_THREADS: '8' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-1p1d-dep8-prefill-c128-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + safetensors-load-strategy: prefetch + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 4 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 256 + max-num-batched-tokens: 8192 + long-prefill-token-threshold: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + decode: + nodes: 4 + workers: 1 + gpus: 16 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-1p1d-dep32-decode-c128-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + safetensors-load-strategy: prefetch + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 16 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 4 + max-num-batched-tokens: 16 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 16 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + sbatch_directives: + cpus-per-task: '72' + mem: '0' + benchmark: + env: + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: 'false' + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + +# Derived from PR #2302's DEP8/DEP32 c256 recipe. +# Halve the prefill/decode topology while preserving its conservative +# 8/32/32 decode limits and isolated JIT cache paths. +override_disagg_1p1d_dep4_dep16_c256_mtp: + name: svf-vllm-disagg-gb300-1p1d-dep4-dep16-c256-mtp-agentic + # GB300 AgentX MTP3 topology: one DEP4 prefill worker feeds one + # and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. + # DEP16 decode worker at concurrency 256. Decode consumes P/D KV through NIXL + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + model: + repo: deepseek-ai/DeepSeek-V4-Pro + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.4.0 + dynamo: + source: + wheel: 1.4.0 + environment: + # Mooncake prefix-block hashes must match across processes and nodes. + PYTHONHASHSEED: '0' + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 2160 + resources: + het_jobs: false + spread_workers: false + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 180GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: random + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + DYN_TCP_CHANNEL_BUFFER: '128' + DYN_TCP_REQUEST_TIMEOUT: '60' + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_MOONCAKE_STORE_SEND_THREADS: '8' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-1p1d-dep8-prefill-c256-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + safetensors-load-strategy: prefetch + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 4 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 256 + max-num-batched-tokens: 8192 + long-prefill-token-threshold: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + decode: + nodes: 4 + workers: 1 + gpus: 16 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-1p1d-dep32-decode-c256-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + safetensors-load-strategy: prefetch + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 16 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 8 + max-num-batched-tokens: 32 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 32 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + sbatch_directives: + cpus-per-task: '72' + mem: '0' + benchmark: + env: + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: 'false' + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + +override_disagg_1p1d_dep4_dep8_c256_mtp: + name: svf-vllm-disagg-gb300-1p1d-dep4-dep8-c256-mtp-agentic + # GB300 AgentX MTP3 topology: one DEP4 prefill worker feeds one + # DEP8 decode worker at concurrency 256. + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + model: + repo: deepseek-ai/DeepSeek-V4-Pro + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.3.0.dev20260720 + dynamo: + source: + wheel: 1.3.0.dev20260720 + environment: + # Mooncake prefix-block hashes must match across processes and nodes. + PYTHONHASHSEED: '0' + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 2160 + resources: + het_jobs: false + spread_workers: false + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 180GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: random + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + DYN_TCP_CHANNEL_BUFFER: '128' + DYN_TCP_REQUEST_TIMEOUT: '60' + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-1p1d-dep4-dep8-c256-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 4 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 8192 + long-prefill-token-threshold: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + max-cudagraph-capture-size: 256 + gpu-memory-utilization: 0.92 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + decode: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-1p1d-dep4-dep8-c256-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 256 + max-num-batched-tokens: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 1024 + gpu-memory-utilization: 0.92 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + sbatch_directives: + cpus-per-task: '72' + benchmark: + env: + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: 'false' + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + +# Source: https://github.com/SemiAnalysisAI/InferenceX/blob/0c33d4615792705ed12bfc204e3a54cfa436cf02/benchmarks/multi_node/srt-slurm-recipes/vllm/deepseek-v4/agentic/disagg-gb300-1p1d-dep8-dep16-c512-mtp-agentic.yaml +# Runtime and performance arguments follow the current DEP8/DEP16 GB300 P/D +# recipes; throughput-only synthetic MTP acceptance is injected at launch. +override_disagg_1p1d_dep8_dep16_c512_mtp: + name: svf-vllm-disagg-gb300-1p1d-dep8-dep16-c512-mtp-agentic + # and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. + # GB300 AgentX MTP3 topology: one DEP8 prefill worker feeds one + # DEP16 decode worker at concurrency 512. Decode consumes P/D KV through NIXL + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + model: + repo: deepseek-ai/DeepSeek-V4-Pro + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.4.0 + dynamo: + source: + wheel: 1.4.0 + environment: + # Mooncake prefix-block hashes must match across processes and nodes. + PYTHONHASHSEED: '0' + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 2160 + resources: + het_jobs: false + spread_workers: false + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 180GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: random + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + DYN_TCP_CHANNEL_BUFFER: '128' + DYN_TCP_REQUEST_TIMEOUT: '60' + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_MOONCAKE_STORE_SEND_THREADS: '8' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-1p1d-dep8-prefill-c512-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + safetensors-load-strategy: prefetch + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 16384 + long-prefill-token-threshold: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + decode: + nodes: 4 + workers: 1 + gpus: 16 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-1p1d-dep16-decode-c512-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + safetensors-load-strategy: prefetch + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 16 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 256 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + sbatch_directives: + cpus-per-task: '72' + mem: '0' + benchmark: + env: + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: 'false' + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + +# Source: https://github.com/Inferact/srt-slurm-sa/blob/4a870cd5bc333bf8a312fc65ca1ea82cdab9b2df/recipes/vllm/deepseek-v4-pro/GB300/agentic/agentx-v1.0.1/1p3d-pdep4-dtp8-c3-kv-nixl-mtp-flashinfer-ar-lpt512-psi1.yaml +# Adapted from the source topology to a 1P4D concurrency-4 point for the +# InferenceX AgentX harness; eval-only runs continue to verify real MTP output. +override_disagg_1p4d_dep4_tp8_c4_mtp: + name: svf-vllm-disagg-gb300-1p4d-dep4-tp8-c4-mtp-agentic + # GB300 high-interactivity AgentX MTP3 topology: one DEP4 prefill worker + # feeds four TP8 decode workers at concurrency 4 through NIXL. + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + model: + repo: deepseek-ai/DeepSeek-V4-Pro + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.4.0 + dynamo: + source: + wheel: 1.4.0 + request_plane: tcp + environment: + DYNAMO_WHEEL_DIRS: /srtctl-wheels + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_FLASHINFER_ALLREDUCE_BACKEND: auto + VLLM_USE_NCCL_SYMM_MEM: '0' + TORCH_SYMMMEM: NVSHMEM + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-1p4d-{job_id} + # Multi-node TP8 decode spans two GB300 nodes; match the sibling recipes' + # NCCL/UCX fabric settings (MNNVL/NVLS, IB HCAs). + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + slurm: + time_limit: 08:00:00 + health_check: + max_attempts: 2160 + resources: + het_jobs: false + spread_workers: false + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + args: + router-mode: kv + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + tokenizer: fastokens + env: + MODEL: deepseek-ai/DeepSeek-V4-Pro + MODEL_PREFIX: dsv4 + FRAMEWORK: dynamo-vllm + PRECISION: fp4 + CONC: '4' + DURATION: '3600' + KV_OFFLOADING: none + ETCD_LEASE_TTL: '120' + DYN_ROUTER_QUEUE_THRESHOLD: None + DYN_TOKENIZER_CACHE: '1' + DYN_TOKENIZER_CACHE_BYTES: '8000000000' + DYN_TOKENIZER: fastokens + DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + placement: + node: first_decode + engine: + dp_launch_mode: per_node + # vLLM KV routing needs prefill KV events so Dynamo can select the + # cache-owning PDEP4 rank. + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + args: + kv-transfer-config: '{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + safetensors-load-strategy: prefetch + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 4 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 8192 + long-prefill-token-threshold: 512 + prefill-schedule-interval: 1 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + gpu-memory-utilization: 0.92 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + # MTP capture sizes are tokens: 64 seqs * (1 target + 3 drafts). + max-cudagraph-capture-size: 256 + kv_events: true + decode: + nodes: 8 + workers: 4 + gpus: 8 + args: + kv-transfer-config: '{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}}' + # TP8 spans two GB300 nodes; custom all-reduce is single-node only, so + # decode uses the FlashInfer allreduce path like the agg TP8 recipes. + disable-custom-all-reduce: true + served-model-name: deepseek-ai/DeepSeek-V4-Pro + safetensors-load-strategy: prefetch + kv-cache-dtype: fp8 + tensor-parallel-size: 8 + pipeline-parallel-size: 1 + max-model-len: 1048576 + max-num-seqs: 16 + max-num-batched-tokens: 64 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + # MTP capture sizes are tokens: 16 seqs * (1 target + 3 drafts). + max-cudagraph-capture-size: 64 + gpu-memory-utilization: 0.92 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + sbatch_directives: + cpus-per-task: '72' + mem: '0' + # The AgentX client uses localhost:8000, so colocate it with the Dynamo + # frontend launched on the first decode node. + benchmark: + env: + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: 'false' + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + MODEL: deepseek-ai/DeepSeek-V4-Pro + MODEL_PREFIX: dsv4 + SERVED_MODEL_NAME: deepseek-ai/DeepSeek-V4-Pro + FRAMEWORK: dynamo-vllm + PRECISION: fp4 + CONC: '4' + DURATION: '3600' + RUNNER_TYPE: gb300 + IMAGE: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + SPEC_DECODING: mtp + DISAGG: 'true' + OFFLOADING: none + KV_OFFLOADING: none + TP: '8' + PREFILL_TP: '1' + PREFILL_NUM_WORKERS: '1' + PREFILL_EP: '4' + DECODE_TP: '8' + DECODE_NUM_WORKERS: '4' + DECODE_EP: '1' + EP_SIZE: '1' + AIPERF_MAX_OSL: none + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + NUM_DATASET_ENTRIES: '393' + HF_WEKA_DATASET: semianalysisai/cc-traces-weka-062126 + PUBLIC_DATASET: semianalysis_cc_traces_weka_062126 + placement: + node: first_decode + +# (benchmark.env.AIPERF_DATASET_MMAP_CACHE_DIR) Container-side path of the aiperf mmap dataset cache; the host-side +# (benchmark.env.AIPERF_DATASET_MMAP_CACHE_DIR) mount is wired via launch_gb300-*.sh's srtslurm.yaml default_mounts. +# (benchmark.env.AIPERF_DATASET_MMAP_CACHE_DIR) Without this, aiperf re-tokenizes + re-writes ~65 GB of mmap files +# (benchmark.env.AIPERF_DATASET_MMAP_CACHE_DIR) per dataset on every run. +# (benchmark.env.HF_HUB_CACHE) Persistent HF hub cache (also wired via default_mounts) so the trace +# (benchmark.env.HF_HUB_CACHE) dataset isn't re-downloaded on every run. Overrides the workflow-level +# (benchmark.env.HF_HUB_CACHE) HF_HUB_CACHE=/mnt/hf_hub_cache, which doesn't exist on these nodes. +override_disagg_1p6d_dep4_tp4: + name: svf-vllm-disagg-gb300-1p6d-dep4-tp4-agentic + # Agentic-coding variant of dsv4/vllm/gb300-fp4/8k1k/disagg-1p6d-dep4-tp4-stp.yaml. + # Topology is identical (1 prefill DEP=4 + 6 decode TP=4, 28 GPUs across 7 + # GB300 nodes + 1 dedicated NATS/etcd infra node) so we can compare against + # the fixed-seq-len 1p6d baseline at the same concurrency point (192). + # + # Divergence vs the 8k1k sibling: + # - benchmark.type: sa-bench -> custom (hands off to agentic_srt.sh) + # - max-model-len: removed (let vLLM derive from model config; agentic + # trajectories blow past any small explicit cap) + # - no-enable-prefix-caching: dropped (prefix caching MUST be on for + # trajectory reuse — entire point of agentic) + # Note: --enable-auto-tool-choice / --tool-call-parser / --reasoning-parser + # are NOT set on the worker. The dynamo-vllm worker entrypoint doesn't + # accept them (different arg parser than `vllm serve`). In disagg, chat + # parsing happens at the dynamo frontend, not at the worker. + model: + container: vllm/vllm-openai:v0.21.0-ubuntu2404 + dynamo: + source: + wheel: 1.2.0.dev20260426 + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 1440 + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + TORCH_SYMMMEM: NVSHMEM + args: + kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 4 + data-parallel-rpc-port: 13345 + enable-expert-parallel: true + attention-config: '{"use_fp4_indexer_cache": true}' + moe-backend: deep_gemm_mega_moe + # enforce-eager: true + # max-num-seqs: 256 + max-num-batched-tokens: 16384 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + gpu-memory-utilization: 0.9 + enable-ep-weight-filter: true + no-disable-hybrid-kv-cache-manager: true + enable-sleep-mode: true + tokenizer-mode: deepseek_v4 + decode: + nodes: 6 + workers: 6 + gpus: 4 + env: + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + TORCH_SYMMMEM: NVSHMEM + args: + kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 4 + pipeline-parallel-size: 1 + # max-num-seqs: 512 + trust-remote-code: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + gpu-memory-utilization: 0.9 + stream-interval: 10 + no-disable-hybrid-kv-cache-manager: true + enable-ep-weight-filter: true + all2all-backend: flashinfer_nvlink_one_sided + no-enable-flashinfer-autotune: true + enable-sleep-mode: true + tokenizer-mode: deepseek_v4 + # sbatch + srun resource grants for clusters without per-GPU defaults. + # + # mem=0: allocate all available node memory (~868 GB on CW gb300). Without + # this, sbatch only requests ntasks × DefMemPerCPU = 8 × 4 GB = 32 GB for + # the whole job and worker cgroups OOM-kill mid model load (R7-R11 hit + # this; sacct showed AllocTRES mem=4G per step). + # cpus-per-task=72: give each task one CW gb300 NUMA socket (144 cores + # split 2 × 72). Critical for the *infra step* (etcd + nats) which + # srtctl spawns without --gres=gpu — on CW that means DefMemPerCPU + # applies and the step gets 1 CPU by default. With 24 dynamo DP ranks + # all hammering etcd for lease keep-alives, single-CPU etcd can't keep + # up and dies (R12 hit this; etcd reported max-cpu-set=1, leases + # deadline-exceeded, infra SIGKILL'd at 16:35:49). 72 CPUs is plenty + # for both etcd + nats AND for vLLM worker auxiliary threads. + # nv gb300 doesn't need this because cluster default DefCpuPerGPU=35 + # auto-allocates 4*35=140 CPUs per GPU-bearing task; cw has no per-GPU + # default. Setting it here is safe on both because the value is ≤ node + # CPU count. + # srun_options.mem=0 forces each srun step to use the full node memory + # (without it, srun steps default back to cpus_per_task × DefMemPerCPU). + # Docs: docs/config-reference.md#sbatch_directives + #srun_options. + sbatch_directives: + cpus-per-task: '72' + mem: '0' + # gb300-nv: pyxis maps the calling user (sa-shared) into the container as + # uid 345200007. dpkg refuses to run without EUID 0 even though + # ENROOT_ROOTFS_WRITABLE=1 makes the rootfs writable, so the agentic_srt + # apt-get install git step fails. --container-remap-root asks pyxis to + # remap us to uid 0 inside the container. srt-slurm renders empty-string + # values as flag-only srun args (see core/slurm.py:250). + srun_options: + mem: '0' + +override_disagg_2p1d_dep8_dep12_c1152_mtp: + name: svf-vllm-disagg-gb300-2p1d-dep8-dep12-c1152-mtp-agentic + # and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. + # GB300 AgentX MTP3 topology: two DEP8 prefill workers feed one + # DEP12 decode worker at concurrency 1152. Decode consumes P/D KV through NIXL + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + model: + repo: deepseek-ai/DeepSeek-V4-Pro + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.3.0.dev20260720 + dynamo: + source: + wheel: 1.3.0.dev20260720 + environment: + # Mooncake prefix-block hashes must match across processes and nodes. + PYTHONHASHSEED: '0' + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 2160 + resources: + het_jobs: false + spread_workers: false + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 180GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: random + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + DYN_TCP_CHANNEL_BUFFER: '128' + DYN_TCP_REQUEST_TIMEOUT: '60' + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 4 + workers: 2 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_MOONCAKE_STORE_SEND_THREADS: '8' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-2p1d-prefill-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 8192 + long-prefill-token-threshold: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + decode: + nodes: 3 + workers: 1 + gpus: 12 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-2p1d-dep12-decode-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 12 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 256 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + sbatch_directives: + cpus-per-task: '72' + mem: '0' + benchmark: + env: + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: 'false' + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + +override_disagg_2p1d_dep8_dep16_c1024_mtp: + name: svf-vllm-disagg-gb300-2p1d-dep8-dep16-c1024-mtp-agentic + # and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. + # GB300 AgentX MTP3 topology: two DEP8 prefill workers feed one + # DEP16 decode worker at concurrency 1024. Decode consumes P/D KV through NIXL + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + identity: + model: + repo: deepseek-ai/DeepSeek-V4-Pro + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f + frameworks: + dynamo: 1.3.0.dev20260720 + dynamo: + source: + wheel: 1.3.0.dev20260720 + environment: + # Mooncake prefix-block hashes must match across processes and nodes. + PYTHONHASHSEED: '0' + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 2160 + resources: + het_jobs: false + spread_workers: false + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 180GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + args: + router-mode: random + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + DYN_TCP_CHANNEL_BUFFER: '128' + DYN_TCP_REQUEST_TIMEOUT: '60' + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 4 + workers: 2 + gpus: 8 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_MOONCAKE_STORE_SEND_THREADS: '8' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-2p1d-prefill-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 8192 + long-prefill-token-threshold: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + decode: + nodes: 4 + workers: 1 + gpus: 16 + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_V2_WARMUP_MAX_NUM_SEQS: '20' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-gb300-2p1d-dep16-decode-{job_id} + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 16 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 64 + max-num-batched-tokens: 256 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: deep_gemm_amxf4_mega_moe + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + sbatch_directives: + cpus-per-task: '72' + mem: '0' + benchmark: + env: + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: 'false' + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + +# (benchmark.env.AIPERF_DATASET_MMAP_CACHE_DIR) Container-side path of the aiperf mmap dataset cache; the host-side +# (benchmark.env.AIPERF_DATASET_MMAP_CACHE_DIR) mount is wired via launch_gb300-*.sh's srtslurm.yaml default_mounts. +# (benchmark.env.AIPERF_DATASET_MMAP_CACHE_DIR) Without this, aiperf re-tokenizes + re-writes ~65 GB of mmap files +# (benchmark.env.AIPERF_DATASET_MMAP_CACHE_DIR) per dataset on every run. +# (benchmark.env.HF_HUB_CACHE) Persistent HF hub cache (also wired via default_mounts) so the trace +# (benchmark.env.HF_HUB_CACHE) dataset isn't re-downloaded on every run. Overrides the workflow-level +# (benchmark.env.HF_HUB_CACHE) HF_HUB_CACHE=/mnt/hf_hub_cache, which doesn't exist on these nodes. +override_disagg_4p1d_dep4_dep8_24_c4096: + name: svf-vllm-disagg-gb300-4p1d-dep4-dep8-24-c4096-agentic + # + # Divergence vs the 8k1k sibling: + # - benchmark.type: sa-bench -> custom (hands off to agentic_srt.sh) + # - max-model-len: removed (let vLLM derive from model config; agentic + # trajectories blow past any small explicit cap) + # - no-enable-prefix-caching: dropped (prefix caching MUST be on for + # trajectory reuse — entire point of agentic) + # Note: --enable-auto-tool-choice / --tool-call-parser / --reasoning-parser + # are NOT set on the worker. The dynamo-vllm worker entrypoint doesn't + # accept them (different arg parser than `vllm serve`). In disagg, chat + # parsing happens at the dynamo frontend, not at the worker. + # Agentic-coding variant of dsv4/vllm/gb300-fp4/8k1k/disagg-4p1d-dep4-dep8-24-c4096-stp.yaml. + # Max-throughput shape: 4 prefill (DEP=4 each) + 1 decode (DEP=8). 6 GB300 + # nodes (4P + 2D = 24 GPUs at 4 GPUs/node) plus a dedicated NATS/etcd infra + # node. Sized for concurrency 4096 with deep_gemm_mega_moe on both workers. + model: + container: vllm/vllm-openai:v0.21.0-ubuntu2404 + dynamo: + source: + wheel: 1.2.0.dev20260426 + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 1440 + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 + roles: + prefill: + nodes: 4 + workers: 4 + gpus: 4 + env: + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_USE_NCCL_SYMM_MEM: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + TORCH_SYMMMEM: NVSHMEM + args: + kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 4 + data-parallel-rpc-port: 13345 + enable-expert-parallel: true + # enforce-eager: true + # Inherited from fixed-sequence recipes; let vLLM select the scheduler + # sequence limit until this is tuned explicitly for the agentic trace. + # max-num-seqs: 16 + max-num-batched-tokens: 16384 + trust-remote-code: true + no-enable-flashinfer-autotune: true + safetensors-load-strategy: prefetch + block-size: 256 + gpu-memory-utilization: 0.9 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + enable-ep-weight-filter: true + enable-sleep-mode: true + moe-backend: deep_gemm_mega_moe + decode: + nodes: 2 + workers: 1 + gpus: 8 + env: + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_USE_NCCL_SYMM_MEM: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + TORCH_SYMMMEM: NVSHMEM + args: + kv-transfer-config: '{"kv_connector": "NixlConnector", "kv_role": "kv_both"}' + served-model-name: deepseek-ai/DeepSeek-V4-Pro + kv-cache-dtype: fp8 + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 8 + data-parallel-rpc-port: 13345 + enable-expert-parallel: true + # max-num-seqs: 512 + trust-remote-code: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + gpu-memory-utilization: 0.9 + stream-interval: 10 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: deepseek_v4 + enable-ep-weight-filter: true + enable-sleep-mode: true + moe-backend: deep_gemm_mega_moe + # sbatch + srun resource grants for clusters without per-GPU defaults. + # + # mem=0: allocate all available node memory (~868 GB on CW gb300). Without + # this, sbatch only requests ntasks × DefMemPerCPU = 8 × 4 GB = 32 GB for + # the whole job and worker cgroups OOM-kill mid model load (R7-R11 hit + # this; sacct showed AllocTRES mem=4G per step). + # cpus-per-task=72: give each task one CW gb300 NUMA socket (144 cores + # split 2 × 72). Critical for the *infra step* (etcd + nats) which + # srtctl spawns without --gres=gpu — on CW that means DefMemPerCPU + # applies and the step gets 1 CPU by default. With 24 dynamo DP ranks + # all hammering etcd for lease keep-alives, single-CPU etcd can't keep + # up and dies (R12 hit this; etcd reported max-cpu-set=1, leases + # deadline-exceeded, infra SIGKILL'd at 16:35:49). 72 CPUs is plenty + # for both etcd + nats AND for vLLM worker auxiliary threads. + # nv gb300 doesn't need this because cluster default DefCpuPerGPU=35 + # auto-allocates 4*35=140 CPUs per GPU-bearing task; cw has no per-GPU + # default. Setting it here is safe on both because the value is ≤ node + # CPU count. + # srun_options.mem=0 forces each srun step to use the full node memory + # (without it, srun steps default back to cpus_per_task × DefMemPerCPU). + # Docs: docs/config-reference.md#sbatch_directives + #srun_options. + sbatch_directives: + cpus-per-task: '72' + mem: '0' + # gb300-nv: pyxis maps the calling user (sa-shared) into the container as + # uid 345200007. dpkg refuses to run without EUID 0 even though + # ENROOT_ROOTFS_WRITABLE=1 makes the rootfs writable, so the agentic_srt + # apt-get install git step fails. --container-remap-root asks pyxis to + # remap us to uid 0 inside the container. srt-slurm renders empty-string + # values as flag-only srun args (see core/slurm.py:250). + srun_options: + mem: '0' diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c1-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c1-mtp.yaml deleted file mode 100644 index 92cd251488..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c1-mtp.yaml +++ /dev/null @@ -1,124 +0,0 @@ -# Agentic-coding SGLang aggregated recipe for GLM-5.2-NVFP4 on B200 at concurrency 1. -# -# One TP8 worker serves prefill and decode on a single node. The sweep matrix -# supplies the concurrency list; agentic_srt.sh replays every point against -# this one server. Acceptance is pinned to the golden thinking-on AL for three -# speculative tokens (golden_al_distribution/glm5.2_mtp.yaml). -schema: 2 -name: agg-b200-tp8-c1-mtp -model: - path: glm-5.2-fp4 - container: dynamo-sglang - precision: fp4 -identity: - model: - repo: nvidia/GLM-5.2-NVFP4 - revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa - container: - image: lmsysorg/sglang:nightly-dev-20260910-00840301 - frameworks: - dynamo: 1.5.0.dev20260909 - sglang: 0.0.0.dev1+g008403017 -resources: - gpu_type: b200 - gpus_per_node: 8 -dynamo: - install: false -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 8 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_REASONING_EFFORT: max - SGLANG_HICACHE_DEBUG_LOG: '1' - SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - max-running-requests: 2 - cuda-graph-max-bs-decode: 2 - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - context-length: 1048576 - speculative-algorithm: EAGLE - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - dsa-prefill-backend: trtllm - dsa-decode-backend: trtllm - moe-runner-backend: flashinfer_trtllm - fp4-gemm-backend: flashinfer_trtllm - disable-shared-experts-fusion: true - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.8 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-size: 110 - hicache-io-backend: direct - watchdog-timeout: 1800 - enable-metrics: true - enable-cache-report: true -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache -health_check: - max_attempts: 1440 - interval_seconds: 10 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 64 -frontend: - type: dynamo - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - router-session-affinity-ttl-secs: 3600 -sbatch_directives: - mem: '0' -srun_options: - mem: '0' - container-remap-root: '' diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c4-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c4-mtp.yaml deleted file mode 100644 index d28d641d54..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c4-mtp.yaml +++ /dev/null @@ -1,124 +0,0 @@ -# Agentic-coding SGLang aggregated recipe for GLM-5.2-NVFP4 on B200 at concurrency 4. -# -# One TP8 worker serves prefill and decode on a single node. The sweep matrix -# supplies the concurrency list; agentic_srt.sh replays every point against -# this one server. Acceptance is pinned to the golden thinking-on AL for three -# speculative tokens (golden_al_distribution/glm5.2_mtp.yaml). -schema: 2 -name: agg-b200-tp8-c4-mtp -model: - path: glm-5.2-fp4 - container: dynamo-sglang - precision: fp4 -identity: - model: - repo: nvidia/GLM-5.2-NVFP4 - revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa - container: - image: lmsysorg/sglang:nightly-dev-20260910-00840301 - frameworks: - dynamo: 1.5.0.dev20260909 - sglang: 0.0.0.dev1+g008403017 -resources: - gpu_type: b200 - gpus_per_node: 8 -dynamo: - install: false -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 8 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_REASONING_EFFORT: max - SGLANG_HICACHE_DEBUG_LOG: '1' - SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - max-running-requests: 8 - cuda-graph-max-bs-decode: 8 - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - context-length: 1048576 - speculative-algorithm: EAGLE - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - dsa-prefill-backend: trtllm - dsa-decode-backend: trtllm - moe-runner-backend: flashinfer_trtllm - fp4-gemm-backend: flashinfer_trtllm - disable-shared-experts-fusion: true - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.8 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-size: 110 - hicache-io-backend: direct - watchdog-timeout: 1800 - enable-metrics: true - enable-cache-report: true -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache -health_check: - max_attempts: 1440 - interval_seconds: 10 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 64 -frontend: - type: dynamo - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - router-session-affinity-ttl-secs: 3600 -sbatch_directives: - mem: '0' -srun_options: - mem: '0' - container-remap-root: '' diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c8-mtp.yaml deleted file mode 100644 index 6b89ba241b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c8-mtp.yaml +++ /dev/null @@ -1,124 +0,0 @@ -# Agentic-coding SGLang aggregated recipe for GLM-5.2-NVFP4 on B200 at concurrency 8. -# -# One TP8 worker serves prefill and decode on a single node. The sweep matrix -# supplies the concurrency list; agentic_srt.sh replays every point against -# this one server. Acceptance is pinned to the golden thinking-on AL for three -# speculative tokens (golden_al_distribution/glm5.2_mtp.yaml). -schema: 2 -name: agg-b200-tp8-c8-mtp -model: - path: glm-5.2-fp4 - container: dynamo-sglang - precision: fp4 -identity: - model: - repo: nvidia/GLM-5.2-NVFP4 - revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa - container: - image: lmsysorg/sglang:nightly-dev-20260910-00840301 - frameworks: - dynamo: 1.5.0.dev20260909 - sglang: 0.0.0.dev1+g008403017 -resources: - gpu_type: b200 - gpus_per_node: 8 -dynamo: - install: false -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 8 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_REASONING_EFFORT: max - SGLANG_HICACHE_DEBUG_LOG: '1' - SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - max-running-requests: 16 - cuda-graph-max-bs-decode: 16 - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - context-length: 1048576 - speculative-algorithm: EAGLE - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - dsa-prefill-backend: trtllm - dsa-decode-backend: trtllm - moe-runner-backend: flashinfer_trtllm - fp4-gemm-backend: flashinfer_trtllm - disable-shared-experts-fusion: true - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.8 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-size: 110 - hicache-io-backend: direct - watchdog-timeout: 1800 - enable-metrics: true - enable-cache-report: true -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache -health_check: - max_attempts: 1440 - interval_seconds: 10 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 64 -frontend: - type: dynamo - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - router-session-affinity-ttl-secs: 3600 -sbatch_directives: - mem: '0' -srun_options: - mem: '0' - container-remap-root: '' diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp.yaml deleted file mode 100644 index dd875b3708..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp.yaml +++ /dev/null @@ -1,198 +0,0 @@ -schema: 2 -name: disagg-b200-1p1d-dep8-dep8-c64-mtp -model: - path: glm-5.2-fp4 - container: dynamo-sglang - precision: fp4 -identity: - model: - repo: nvidia/GLM-5.2-NVFP4 - revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa - container: - image: lmsysorg/sglang:nightly-dev-20260910-00840301 - frameworks: - dynamo: 1.5.0.dev20260909 - sglang: 0.0.0.dev1+g008403017 -resources: - gpu_type: b200 - gpus_per_node: 8 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None -dynamo: - install: false -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 8 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_ENGINE_INIT_TIMEOUT: '1800' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_HICACHE_DEBUG_LOG: '1' - SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' - SGLANG_REASONING_EFFORT: max - PIP_BREAK_SYSTEM_PACKAGES: '1' - NVSHMEM_REMOTE_TRANSPORT: none - SGLANG_DG_CACHE_DIR: /deepgemm_cache - FLASHINFER_WORKSPACE_BASE: /flashinfer_cache - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: total_tokens - chunked-prefill-size: 65536 - max-prefill-tokens: 8192 - max-running-requests: 16 - cuda-graph-max-bs-decode: 16 - dsa-prefill-backend: trtllm - dsa-decode-backend: trtllm - moe-runner-backend: flashinfer_cutlass - fp4-gemm-backend: flashinfer_cutlass - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.8 - context-length: 1048576 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-size: 110 - hicache-io-backend: direct - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - watchdog-timeout: 1800 - enable-metrics: true - enable-cache-report: true - decode: - nodes: 1 - workers: 1 - gpus: 8 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_ENGINE_INIT_TIMEOUT: '1800' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - SGLANG_REASONING_EFFORT: max - PIP_BREAK_SYSTEM_PACKAGES: '1' - NVSHMEM_REMOTE_TRANSPORT: none - SGLANG_DG_CACHE_DIR: /deepgemm_cache - FLASHINFER_WORKSPACE_BASE: /flashinfer_cache - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - disable-radix-cache: true - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - max-running-requests: 128 - cuda-graph-max-bs-decode: 128 - chunked-prefill-size: 64 - context-length: 1048576 - dsa-prefill-backend: trtllm - dsa-decode-backend: trtllm - moe-runner-backend: flashinfer_cutedsl - fp4-gemm-backend: flashinfer_cutlass - skip-tokenizer-init: true - stream-interval: 30 - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.85 - disaggregation-decode-extra-slots: 0 - watchdog-timeout: 1800 - enable-metrics: true - enable-cache-report: true - deepep-config: /configs/deepep_config.json - deepep-mode: low_latency - ep-dispatch-algorithm: static - ep-num-redundant-experts: 0 - moe-a2a-backend: deepep - moe-dense-tp-size: 1 - speculative-moe-a2a-backend: deepep - speculative-moe-runner-backend: deep_gemm -health_check: - max_attempts: 1440 - interval_seconds: 10 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 64 -sbatch_directives: - mem: '0' -srun_options: - mem: '0' - container-remap-root: '' diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/disagg-1p4d-dep8-tp4-c48-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/disagg-1p4d-dep8-tp4-c48-mtp.yaml deleted file mode 100644 index ac814c39a1..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/disagg-1p4d-dep8-tp4-c48-mtp.yaml +++ /dev/null @@ -1,190 +0,0 @@ -schema: 2 -name: disagg-b200-1p4d-dep8-tp4-c48-mtp -model: - path: glm-5.2-fp4 - container: dynamo-sglang - precision: fp4 -identity: - model: - repo: nvidia/GLM-5.2-NVFP4 - revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa - container: - image: lmsysorg/sglang:nightly-dev-20260910-00840301 - frameworks: - dynamo: 1.5.0.dev20260909 - sglang: 0.0.0.dev1+g008403017 -resources: - gpu_type: b200 - gpus_per_node: 8 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None -dynamo: - install: false -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 8 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_ENGINE_INIT_TIMEOUT: '3600' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_HICACHE_DEBUG_LOG: '1' - SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' - SGLANG_REASONING_EFFORT: max - PIP_BREAK_SYSTEM_PACKAGES: '1' - NVSHMEM_REMOTE_TRANSPORT: none - SGLANG_DG_CACHE_DIR: /deepgemm_cache - FLASHINFER_WORKSPACE_BASE: /flashinfer_cache - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: total_tokens - chunked-prefill-size: 65536 - max-prefill-tokens: 16384 - max-running-requests: 16 - cuda-graph-max-bs-decode: 16 - dsa-prefill-backend: trtllm - dsa-decode-backend: trtllm - moe-runner-backend: flashinfer_cutlass - fp4-gemm-backend: flashinfer_cutlass - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.8 - context-length: 1048576 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-size: 110 - hicache-io-backend: direct - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - watchdog-timeout: 1800 - enable-metrics: true - enable-cache-report: true - decode: - nodes: 2 - workers: 4 - gpus: 4 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_ENGINE_INIT_TIMEOUT: '3600' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - SGLANG_REASONING_EFFORT: max - PIP_BREAK_SYSTEM_PACKAGES: '1' - NVSHMEM_REMOTE_TRANSPORT: none - SGLANG_DG_CACHE_DIR: /deepgemm_cache - FLASHINFER_WORKSPACE_BASE: /flashinfer_cache - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - disable-radix-cache: true - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - max-running-requests: 16 - cuda-graph-max-bs-decode: 16 - chunked-prefill-size: 64 - context-length: 1048576 - dsa-prefill-backend: trtllm - dsa-decode-backend: trtllm - moe-runner-backend: flashinfer_trtllm - fp4-gemm-backend: flashinfer_cutlass - skip-tokenizer-init: true - stream-interval: 30 - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.9 - disaggregation-decode-extra-slots: 0 - watchdog-timeout: 1800 - enable-metrics: true - enable-cache-report: true -health_check: - max_attempts: 1440 - interval_seconds: 10 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 64 -sbatch_directives: - mem: '0' -srun_options: - mem: '0' - container-remap-root: '' diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..9691c961b9 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml @@ -0,0 +1,551 @@ +# srt-slurm recipes for glm5.2/sglang/b200-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: glm-5.2-fp4 + container: dynamo-sglang + precision: fp4 + identity: + model: + repo: nvidia/GLM-5.2-NVFP4 + revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa + container: + image: lmsysorg/sglang:nightly-dev-20260910-00840301 + frameworks: + dynamo: 1.5.0.dev20260909 + sglang: 0.0.0.dev1+g008403017 + resources: + gpu_type: b200 + gpus_per_node: 8 + dynamo: + install: false + engine: sglang + roles: {} + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + health_check: + max_attempts: 1440 + interval_seconds: 10 + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 64 + frontend: + type: dynamo + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + router-mode: kv + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + sbatch_directives: + mem: '0' + srun_options: + mem: '0' + container-remap-root: '' + +# Agentic-coding SGLang aggregated recipe for GLM-5.2-NVFP4 on B200 at concurrency 1. +# +# One TP8 worker serves prefill and decode on a single node. The sweep matrix +# supplies the concurrency list; agentic_srt.sh replays every point against +# this one server. Acceptance is pinned to the golden thinking-on AL for three +# speculative tokens (golden_al_distribution/glm5.2_mtp.yaml). +override_agg_tp8_c1_mtp: + name: agg-b200-tp8-c1-mtp + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_REASONING_EFFORT: max + SGLANG_HICACHE_DEBUG_LOG: '1' + SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 8 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + max-running-requests: 2 + cuda-graph-max-bs-decode: 2 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + context-length: 1048576 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + dsa-prefill-backend: trtllm + dsa-decode-backend: trtllm + moe-runner-backend: flashinfer_trtllm + fp4-gemm-backend: flashinfer_trtllm + disable-shared-experts-fusion: true + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.8 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-size: 110 + hicache-io-backend: direct + watchdog-timeout: 1800 + enable-metrics: true + enable-cache-report: true + frontend: + args: + router-session-affinity-ttl-secs: 3600 + +# +# One TP8 worker serves prefill and decode on a single node. The sweep matrix +# supplies the concurrency list; agentic_srt.sh replays every point against +# this one server. Acceptance is pinned to the golden thinking-on AL for three +# speculative tokens (golden_al_distribution/glm5.2_mtp.yaml). +# Agentic-coding SGLang aggregated recipe for GLM-5.2-NVFP4 on B200 at concurrency 4. +override_agg_tp8_c4_mtp: + name: agg-b200-tp8-c4-mtp + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_REASONING_EFFORT: max + SGLANG_HICACHE_DEBUG_LOG: '1' + SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 8 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + max-running-requests: 8 + cuda-graph-max-bs-decode: 8 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + context-length: 1048576 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + dsa-prefill-backend: trtllm + dsa-decode-backend: trtllm + moe-runner-backend: flashinfer_trtllm + fp4-gemm-backend: flashinfer_trtllm + disable-shared-experts-fusion: true + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.8 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-size: 110 + hicache-io-backend: direct + watchdog-timeout: 1800 + enable-metrics: true + enable-cache-report: true + frontend: + args: + router-session-affinity-ttl-secs: 3600 + +# +# One TP8 worker serves prefill and decode on a single node. The sweep matrix +# supplies the concurrency list; agentic_srt.sh replays every point against +# this one server. Acceptance is pinned to the golden thinking-on AL for three +# speculative tokens (golden_al_distribution/glm5.2_mtp.yaml). +# Agentic-coding SGLang aggregated recipe for GLM-5.2-NVFP4 on B200 at concurrency 8. +override_agg_tp8_c8_mtp: + name: agg-b200-tp8-c8-mtp + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_REASONING_EFFORT: max + SGLANG_HICACHE_DEBUG_LOG: '1' + SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 8 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + max-running-requests: 16 + cuda-graph-max-bs-decode: 16 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + context-length: 1048576 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + dsa-prefill-backend: trtllm + dsa-decode-backend: trtllm + moe-runner-backend: flashinfer_trtllm + fp4-gemm-backend: flashinfer_trtllm + disable-shared-experts-fusion: true + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.8 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-size: 110 + hicache-io-backend: direct + watchdog-timeout: 1800 + enable-metrics: true + enable-cache-report: true + frontend: + args: + router-session-affinity-ttl-secs: 3600 + +override_disagg_1p1d_dep8_dep8_c64_mtp: + name: disagg-b200-1p1d-dep8-dep8-c64-mtp + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 8 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_ENGINE_INIT_TIMEOUT: '1800' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_HICACHE_DEBUG_LOG: '1' + SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' + SGLANG_REASONING_EFFORT: max + PIP_BREAK_SYSTEM_PACKAGES: '1' + NVSHMEM_REMOTE_TRANSPORT: none + SGLANG_DG_CACHE_DIR: /deepgemm_cache + FLASHINFER_WORKSPACE_BASE: /flashinfer_cache + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + disaggregation-mode: prefill + disaggregation-transfer-backend: nixl + tensor-parallel-size: 8 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: total_tokens + chunked-prefill-size: 65536 + max-prefill-tokens: 8192 + max-running-requests: 16 + cuda-graph-max-bs-decode: 16 + dsa-prefill-backend: trtllm + dsa-decode-backend: trtllm + moe-runner-backend: flashinfer_cutlass + fp4-gemm-backend: flashinfer_cutlass + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.8 + context-length: 1048576 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-size: 110 + hicache-io-backend: direct + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + watchdog-timeout: 1800 + enable-metrics: true + enable-cache-report: true + decode: + nodes: 1 + workers: 1 + gpus: 8 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_ENGINE_INIT_TIMEOUT: '1800' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + SGLANG_REASONING_EFFORT: max + PIP_BREAK_SYSTEM_PACKAGES: '1' + NVSHMEM_REMOTE_TRANSPORT: none + SGLANG_DG_CACHE_DIR: /deepgemm_cache + FLASHINFER_WORKSPACE_BASE: /flashinfer_cache + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + disaggregation-mode: decode + disaggregation-transfer-backend: nixl + disable-radix-cache: true + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + tensor-parallel-size: 8 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + max-running-requests: 128 + cuda-graph-max-bs-decode: 128 + chunked-prefill-size: 64 + context-length: 1048576 + dsa-prefill-backend: trtllm + dsa-decode-backend: trtllm + moe-runner-backend: flashinfer_cutedsl + fp4-gemm-backend: flashinfer_cutlass + skip-tokenizer-init: true + stream-interval: 30 + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.85 + disaggregation-decode-extra-slots: 0 + watchdog-timeout: 1800 + enable-metrics: true + enable-cache-report: true + deepep-config: /configs/deepep_config.json + deepep-mode: low_latency + ep-dispatch-algorithm: static + ep-num-redundant-experts: 0 + moe-a2a-backend: deepep + moe-dense-tp-size: 1 + speculative-moe-a2a-backend: deepep + speculative-moe-runner-backend: deep_gemm + benchmark: + env: + AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + frontend: + args: + router-session-affinity-ttl-secs: '3600' + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: true + +override_disagg_1p4d_dep8_tp4_c48_mtp: + name: disagg-b200-1p4d-dep8-tp4-c48-mtp + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 8 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_ENGINE_INIT_TIMEOUT: '3600' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_HICACHE_DEBUG_LOG: '1' + SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' + SGLANG_REASONING_EFFORT: max + PIP_BREAK_SYSTEM_PACKAGES: '1' + NVSHMEM_REMOTE_TRANSPORT: none + SGLANG_DG_CACHE_DIR: /deepgemm_cache + FLASHINFER_WORKSPACE_BASE: /flashinfer_cache + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + disaggregation-mode: prefill + disaggregation-transfer-backend: nixl + tensor-parallel-size: 8 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: total_tokens + chunked-prefill-size: 65536 + max-prefill-tokens: 16384 + max-running-requests: 16 + cuda-graph-max-bs-decode: 16 + dsa-prefill-backend: trtllm + dsa-decode-backend: trtllm + moe-runner-backend: flashinfer_cutlass + fp4-gemm-backend: flashinfer_cutlass + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.8 + context-length: 1048576 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-size: 110 + hicache-io-backend: direct + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + watchdog-timeout: 1800 + enable-metrics: true + enable-cache-report: true + decode: + nodes: 2 + workers: 4 + gpus: 4 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_ENGINE_INIT_TIMEOUT: '3600' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + SGLANG_REASONING_EFFORT: max + PIP_BREAK_SYSTEM_PACKAGES: '1' + NVSHMEM_REMOTE_TRANSPORT: none + SGLANG_DG_CACHE_DIR: /deepgemm_cache + FLASHINFER_WORKSPACE_BASE: /flashinfer_cache + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + disaggregation-mode: decode + disaggregation-transfer-backend: nixl + disable-radix-cache: true + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + max-running-requests: 16 + cuda-graph-max-bs-decode: 16 + chunked-prefill-size: 64 + context-length: 1048576 + dsa-prefill-backend: trtllm + dsa-decode-backend: trtllm + moe-runner-backend: flashinfer_trtllm + fp4-gemm-backend: flashinfer_cutlass + skip-tokenizer-init: true + stream-interval: 30 + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.9 + disaggregation-decode-extra-slots: 0 + watchdog-timeout: 1800 + enable-metrics: true + enable-cache-report: true + benchmark: + env: + AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + frontend: + args: + router-session-affinity-ttl-secs: '3600' + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: true diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c2-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c2-mtp.yaml deleted file mode 100644 index c56e03e9be..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c2-mtp.yaml +++ /dev/null @@ -1,128 +0,0 @@ -schema: 2 -name: agg-gb200-tp8-c2-mtp -model: - path: glm-5.2-fp4 - container: dynamo-sglang - precision: fp4 -identity: - model: - repo: nvidia/GLM-5.2-NVFP4 - revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260805-211ee642 - frameworks: - dynamo: 71eb001e17fa73c742f0afe1a6ed96836cb135fd - sglang: nightly-dev-cu13-20260805-211ee642 -resources: - gpu_type: gb200 - gpus_per_node: 4 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None -dynamo: - install: true - source: - rev: 71eb001e17fa73c742f0afe1a6ed96836cb135fd -engine: sglang -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - MC_TE_METRIC: 'true' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_ENABLE_THINKING: '1' - SGLANG_REASONING_EFFORT: max - SGLANG_ENABLE_UNIFIED_RADIX_TREE: '1' - SGLANG_HICACHE_DEBUG_LOG: '1' - SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - max-running-requests: 10 - cuda-graph-max-bs: 10 - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - context-length: 1048576 - speculative-algorithm: EAGLE - speculative-num-steps: 4 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 5 - nsa-prefill-backend: trtllm - nsa-decode-backend: trtllm - moe-runner-backend: flashinfer_trtllm - fp4-gemm-backend: flashinfer_trtllm - disable-shared-experts-fusion: true - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.8 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-size: 135 - hicache-io-backend: direct - enable-metrics: true -health_check: - max_attempts: 1440 - interval_seconds: 10 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 64 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c4-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c4-mtp.yaml deleted file mode 100644 index d845ec11e4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c4-mtp.yaml +++ /dev/null @@ -1,128 +0,0 @@ -schema: 2 -name: agg-gb200-tp8-c4-mtp -model: - path: glm-5.2-fp4 - container: dynamo-sglang - precision: fp4 -identity: - model: - repo: nvidia/GLM-5.2-NVFP4 - revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260805-211ee642 - frameworks: - dynamo: 71eb001e17fa73c742f0afe1a6ed96836cb135fd - sglang: nightly-dev-cu13-20260805-211ee642 -resources: - gpu_type: gb200 - gpus_per_node: 4 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None -dynamo: - install: true - source: - rev: 71eb001e17fa73c742f0afe1a6ed96836cb135fd -engine: sglang -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - MC_TE_METRIC: 'true' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_ENABLE_THINKING: '1' - SGLANG_REASONING_EFFORT: max - SGLANG_ENABLE_UNIFIED_RADIX_TREE: '1' - SGLANG_HICACHE_DEBUG_LOG: '1' - SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - max-running-requests: 10 - cuda-graph-max-bs: 10 - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - context-length: 1048576 - speculative-algorithm: EAGLE - speculative-num-steps: 4 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 5 - nsa-prefill-backend: trtllm - nsa-decode-backend: trtllm - moe-runner-backend: flashinfer_trtllm - fp4-gemm-backend: flashinfer_trtllm - disable-shared-experts-fusion: true - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.8 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-size: 135 - hicache-io-backend: direct - enable-metrics: true -health_check: - max_attempts: 1440 - interval_seconds: 10 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 64 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c8-mtp.yaml deleted file mode 100644 index 6b45a9c09e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c8-mtp.yaml +++ /dev/null @@ -1,128 +0,0 @@ -schema: 2 -name: agg-gb200-tp8-c8-mtp -model: - path: glm-5.2-fp4 - container: dynamo-sglang - precision: fp4 -identity: - model: - repo: nvidia/GLM-5.2-NVFP4 - revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260805-211ee642 - frameworks: - dynamo: 71eb001e17fa73c742f0afe1a6ed96836cb135fd - sglang: nightly-dev-cu13-20260805-211ee642 -resources: - gpu_type: gb200 - gpus_per_node: 4 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None -dynamo: - install: true - source: - rev: 71eb001e17fa73c742f0afe1a6ed96836cb135fd -engine: sglang -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - MC_TE_METRIC: 'true' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_ENABLE_THINKING: '1' - SGLANG_REASONING_EFFORT: max - SGLANG_ENABLE_UNIFIED_RADIX_TREE: '1' - SGLANG_HICACHE_DEBUG_LOG: '1' - SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - max-running-requests: 10 - cuda-graph-max-bs: 10 - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - context-length: 1048576 - speculative-algorithm: EAGLE - speculative-num-steps: 4 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 5 - nsa-prefill-backend: trtllm - nsa-decode-backend: trtllm - moe-runner-backend: flashinfer_trtllm - fp4-gemm-backend: flashinfer_trtllm - disable-shared-experts-fusion: true - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.8 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-size: 135 - hicache-io-backend: direct - enable-metrics: true -health_check: - max_attempts: 1440 - interval_seconds: 10 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 64 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-1p4d-dep8-tp4-c48-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-1p4d-dep8-tp4-c48-mtp.yaml deleted file mode 100644 index 27721263a6..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-1p4d-dep8-tp4-c48-mtp.yaml +++ /dev/null @@ -1,199 +0,0 @@ -schema: 2 -name: disagg-gb200-1p4d-dep8-tp4-c48-mtp -model: - path: glm-5.2-fp4 - container: dynamo-sglang - precision: fp4 -identity: - model: - repo: nvidia/GLM-5.2-NVFP4 - revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260805-211ee642 - frameworks: - dynamo: 71eb001e17fa73c742f0afe1a6ed96836cb135fd - sglang: nightly-dev-cu13-20260805-211ee642 -resources: - gpu_type: gb200 - gpus_per_node: 4 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None -dynamo: - install: true - source: - rev: 71eb001e17fa73c742f0afe1a6ed96836cb135fd -engine: sglang -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - MC_TE_METRIC: 'true' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_ENABLE_THINKING: '1' - SGLANG_ENABLE_UNIFIED_RADIX_TREE: '1' - SGLANG_HICACHE_DEBUG_LOG: '1' - SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' - SGLANG_REASONING_EFFORT: max - PIP_BREAK_SYSTEM_PACKAGES: '1' - UCX_TLS: cuda_copy,cuda_ipc,sm,self,tcp - NVSHMEM_REMOTE_TRANSPORT: none - SGLANG_DG_CACHE_DIR: /deepgemm_cache - FLASHINFER_WORKSPACE_BASE: /flashinfer_cache - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: total_tokens - chunked-prefill-size: 65536 - max-prefill-tokens: 16384 - max-running-requests: 16 - cuda-graph-max-bs: 16 - disable-cuda-graph: true - nsa-prefill-backend: trtllm - nsa-decode-backend: trtllm - moe-runner-backend: flashinfer_cutlass - fp4-gemm-backend: flashinfer_cutlass - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.8 - context-length: 1048576 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-size: 135 - hicache-io-backend: direct - speculative-algorithm: EAGLE - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 2 - enable-metrics: true - enable-cache-report: true - decode: - nodes: 4 - workers: 4 - gpus: 4 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - MC_TE_METRIC: 'true' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_ENABLE_THINKING: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - SGLANG_REASONING_EFFORT: max - PIP_BREAK_SYSTEM_PACKAGES: '1' - UCX_TLS: cuda_copy,cuda_ipc,sm,self,tcp - NVSHMEM_REMOTE_TRANSPORT: none - SGLANG_DG_CACHE_DIR: /deepgemm_cache - FLASHINFER_WORKSPACE_BASE: /flashinfer_cache - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - disable-radix-cache: true - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - max-running-requests: 16 - cuda-graph-max-bs: 16 - chunked-prefill-size: 64 - context-length: 1048576 - nsa-prefill-backend: trtllm - nsa-decode-backend: trtllm - moe-runner-backend: flashinfer_trtllm - fp4-gemm-backend: flashinfer_cutlass - skip-tokenizer-init: true - stream-interval: 30 - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.9 - disaggregation-decode-extra-slots: 0 - enable-metrics: true - enable-cache-report: true -health_check: - max_attempts: 1440 - interval_seconds: 10 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 64 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-1p6d-dep8-tp4-c45-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-1p6d-dep8-tp4-c45-mtp.yaml deleted file mode 100644 index 51f260e38c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-1p6d-dep8-tp4-c45-mtp.yaml +++ /dev/null @@ -1,199 +0,0 @@ -schema: 2 -name: disagg-gb200-1p6d-dep8-tp4-c45-mtp -model: - path: glm-5.2-fp4 - container: dynamo-sglang - precision: fp4 -identity: - model: - repo: nvidia/GLM-5.2-NVFP4 - revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260805-211ee642 - frameworks: - dynamo: 71eb001e17fa73c742f0afe1a6ed96836cb135fd - sglang: nightly-dev-cu13-20260805-211ee642 -resources: - gpu_type: gb200 - gpus_per_node: 4 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None -dynamo: - install: true - source: - rev: 71eb001e17fa73c742f0afe1a6ed96836cb135fd -engine: sglang -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - MC_TE_METRIC: 'true' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_ENABLE_THINKING: '1' - SGLANG_ENABLE_UNIFIED_RADIX_TREE: '1' - SGLANG_HICACHE_DEBUG_LOG: '1' - SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' - SGLANG_REASONING_EFFORT: max - PIP_BREAK_SYSTEM_PACKAGES: '1' - UCX_TLS: cuda_copy,cuda_ipc,sm,self,tcp - NVSHMEM_REMOTE_TRANSPORT: none - SGLANG_DG_CACHE_DIR: /deepgemm_cache - FLASHINFER_WORKSPACE_BASE: /flashinfer_cache - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: total_tokens - chunked-prefill-size: 65536 - max-prefill-tokens: 16384 - max-running-requests: 16 - cuda-graph-max-bs: 16 - disable-cuda-graph: true - nsa-prefill-backend: trtllm - nsa-decode-backend: trtllm - moe-runner-backend: flashinfer_cutlass - fp4-gemm-backend: flashinfer_cutlass - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.8 - context-length: 1048576 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-size: 135 - hicache-io-backend: direct - speculative-algorithm: EAGLE - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 2 - enable-metrics: true - enable-cache-report: true - decode: - nodes: 6 - workers: 6 - gpus: 4 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - MC_TE_METRIC: 'true' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_ENABLE_THINKING: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - SGLANG_REASONING_EFFORT: max - PIP_BREAK_SYSTEM_PACKAGES: '1' - UCX_TLS: cuda_copy,cuda_ipc,sm,self,tcp - NVSHMEM_REMOTE_TRANSPORT: none - SGLANG_DG_CACHE_DIR: /deepgemm_cache - FLASHINFER_WORKSPACE_BASE: /flashinfer_cache - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - disable-radix-cache: true - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - max-running-requests: 16 - cuda-graph-max-bs: 16 - chunked-prefill-size: 64 - context-length: 1048576 - nsa-prefill-backend: trtllm - nsa-decode-backend: trtllm - moe-runner-backend: flashinfer_trtllm - fp4-gemm-backend: flashinfer_cutlass - skip-tokenizer-init: true - stream-interval: 30 - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.9 - disaggregation-decode-extra-slots: 0 - enable-metrics: true - enable-cache-report: true -health_check: - max_attempts: 1440 - interval_seconds: 10 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 64 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-2p1d-dep8-dep16-c128-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-2p1d-dep8-dep16-c128-mtp.yaml deleted file mode 100644 index dcb66f55d2..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-2p1d-dep8-dep16-c128-mtp.yaml +++ /dev/null @@ -1,207 +0,0 @@ -schema: 2 -name: disagg-gb200-2p1d-dep8-dep16-c128-mtp -model: - path: glm-5.2-fp4 - container: dynamo-sglang - precision: fp4 -identity: - model: - repo: nvidia/GLM-5.2-NVFP4 - revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260805-211ee642 - frameworks: - dynamo: 71eb001e17fa73c742f0afe1a6ed96836cb135fd - sglang: nightly-dev-cu13-20260805-211ee642 -resources: - gpu_type: gb200 - gpus_per_node: 4 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None -dynamo: - install: true - source: - rev: 71eb001e17fa73c742f0afe1a6ed96836cb135fd -engine: sglang -roles: - prefill: - nodes: 4 - workers: 2 - gpus: 8 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - MC_TE_METRIC: 'true' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_ENABLE_THINKING: '1' - SGLANG_ENABLE_UNIFIED_RADIX_TREE: '1' - SGLANG_HICACHE_DEBUG_LOG: '1' - SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' - SGLANG_REASONING_EFFORT: max - PIP_BREAK_SYSTEM_PACKAGES: '1' - UCX_TLS: cuda_copy,cuda_ipc,sm,self,tcp - NVSHMEM_REMOTE_TRANSPORT: none - SGLANG_DG_CACHE_DIR: /deepgemm_cache - FLASHINFER_WORKSPACE_BASE: /flashinfer_cache - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - disaggregation-mode: prefill - disaggregation-transfer-backend: nixl - tensor-parallel-size: 8 - data-parallel-size: 8 - expert-parallel-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: total_tokens - chunked-prefill-size: 65536 - max-prefill-tokens: 8192 - max-running-requests: 16 - cuda-graph-max-bs: 16 - disable-cuda-graph: true - nsa-prefill-backend: trtllm - nsa-decode-backend: trtllm - moe-runner-backend: flashinfer_cutlass - fp4-gemm-backend: flashinfer_cutlass - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.8 - context-length: 1048576 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-size: 100 - hicache-io-backend: direct - speculative-algorithm: EAGLE - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 2 - enable-metrics: true - enable-cache-report: true - decode: - nodes: 4 - workers: 1 - gpus: 16 - env: - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - PYTHONUNBUFFERED: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - MC_TE_METRIC: 'true' - MC_FORCE_MNNVL: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_ENABLE_THINKING: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - SGLANG_REASONING_EFFORT: max - PIP_BREAK_SYSTEM_PACKAGES: '1' - UCX_TLS: cuda_copy,cuda_ipc,sm,self,tcp - NVSHMEM_REMOTE_TRANSPORT: none - SGLANG_DG_CACHE_DIR: /deepgemm_cache - FLASHINFER_WORKSPACE_BASE: /flashinfer_cache - args: - served-model-name: nvidia/GLM-5.2-NVFP4 - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - disaggregation-mode: decode - disaggregation-transfer-backend: nixl - disable-radix-cache: true - speculative-algorithm: EAGLE - speculative-num-steps: 2 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 3 - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - enable-dp-attention: true - enable-dp-lm-head: true - max-running-requests: 144 - cuda-graph-max-bs: 144 - chunked-prefill-size: 64 - context-length: 1048576 - nsa-prefill-backend: trtllm - nsa-decode-backend: trtllm - moe-runner-backend: flashinfer_cutedsl - fp4-gemm-backend: flashinfer_cutlass - skip-tokenizer-init: true - stream-interval: 30 - enable-flashinfer-allreduce-fusion: true - weight-loader-prefetch-checkpoints: true - model-loader-extra-config: '{"enable_multithread_load": true}' - mem-fraction-static: 0.85 - disaggregation-decode-extra-slots: 0 - enable-metrics: true - enable-cache-report: true - deepep-config: /configs/deepep_config.json - deepep-mode: low_latency - ep-dispatch-algorithm: static - ep-num-redundant-experts: 0 - moe-a2a-backend: deepep - moe-dense-tp-size: 1 - speculative-moe-a2a-backend: deepep - speculative-moe-runner-backend: deep_gemm -health_check: - max_attempts: 1440 - interval_seconds: 10 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 64 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..a1d76aa950 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml @@ -0,0 +1,663 @@ +# srt-slurm recipes for glm5.2/sglang/gb200-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: glm-5.2-fp4 + container: dynamo-sglang + precision: fp4 + identity: + model: + repo: nvidia/GLM-5.2-NVFP4 + revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa + container: + image: lmsysorg/sglang:nightly-dev-cu13-20260805-211ee642 + frameworks: + dynamo: 71eb001e17fa73c742f0afe1a6ed96836cb135fd + sglang: nightly-dev-cu13-20260805-211ee642 + resources: + gpu_type: gb200 + gpus_per_node: 4 + frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: true + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + dynamo: + install: true + source: + rev: 71eb001e17fa73c742f0afe1a6ed96836cb135fd + engine: sglang + roles: {} + health_check: + max_attempts: 1440 + interval_seconds: 10 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 64 + sbatch_directives: + mem: '0' + cpus-per-task: '144' + srun_options: + mem: '0' + container-remap-root: '' + +override_agg_tp8_c2_mtp: + name: agg-gb200-tp8-c2-mtp + roles: + agg: + nodes: 2 + workers: 1 + gpus: 8 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + MC_TE_METRIC: 'true' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_ENABLE_THINKING: '1' + SGLANG_REASONING_EFFORT: max + SGLANG_ENABLE_UNIFIED_RADIX_TREE: '1' + SGLANG_HICACHE_DEBUG_LOG: '1' + SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 8 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + max-running-requests: 10 + cuda-graph-max-bs: 10 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + context-length: 1048576 + speculative-algorithm: EAGLE + speculative-num-steps: 4 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 5 + nsa-prefill-backend: trtllm + nsa-decode-backend: trtllm + moe-runner-backend: flashinfer_trtllm + fp4-gemm-backend: flashinfer_trtllm + disable-shared-experts-fusion: true + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.8 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-size: 135 + hicache-io-backend: direct + enable-metrics: true + +override_agg_tp8_c4_mtp: + name: agg-gb200-tp8-c4-mtp + roles: + agg: + nodes: 2 + workers: 1 + gpus: 8 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + MC_TE_METRIC: 'true' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_ENABLE_THINKING: '1' + SGLANG_REASONING_EFFORT: max + SGLANG_ENABLE_UNIFIED_RADIX_TREE: '1' + SGLANG_HICACHE_DEBUG_LOG: '1' + SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 8 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + max-running-requests: 10 + cuda-graph-max-bs: 10 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + context-length: 1048576 + speculative-algorithm: EAGLE + speculative-num-steps: 4 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 5 + nsa-prefill-backend: trtllm + nsa-decode-backend: trtllm + moe-runner-backend: flashinfer_trtllm + fp4-gemm-backend: flashinfer_trtllm + disable-shared-experts-fusion: true + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.8 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-size: 135 + hicache-io-backend: direct + enable-metrics: true + +override_agg_tp8_c8_mtp: + name: agg-gb200-tp8-c8-mtp + roles: + agg: + nodes: 2 + workers: 1 + gpus: 8 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + MC_TE_METRIC: 'true' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_ENABLE_THINKING: '1' + SGLANG_REASONING_EFFORT: max + SGLANG_ENABLE_UNIFIED_RADIX_TREE: '1' + SGLANG_HICACHE_DEBUG_LOG: '1' + SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 8 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + max-running-requests: 10 + cuda-graph-max-bs: 10 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + context-length: 1048576 + speculative-algorithm: EAGLE + speculative-num-steps: 4 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 5 + nsa-prefill-backend: trtllm + nsa-decode-backend: trtllm + moe-runner-backend: flashinfer_trtllm + fp4-gemm-backend: flashinfer_trtllm + disable-shared-experts-fusion: true + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.8 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-size: 135 + hicache-io-backend: direct + enable-metrics: true + +override_disagg_1p4d_dep8_tp4_c48_mtp: + name: disagg-gb200-1p4d-dep8-tp4-c48-mtp + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_TE_METRIC: 'true' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_ENABLE_THINKING: '1' + SGLANG_ENABLE_UNIFIED_RADIX_TREE: '1' + SGLANG_HICACHE_DEBUG_LOG: '1' + SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' + SGLANG_REASONING_EFFORT: max + PIP_BREAK_SYSTEM_PACKAGES: '1' + UCX_TLS: cuda_copy,cuda_ipc,sm,self,tcp + NVSHMEM_REMOTE_TRANSPORT: none + SGLANG_DG_CACHE_DIR: /deepgemm_cache + FLASHINFER_WORKSPACE_BASE: /flashinfer_cache + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + disaggregation-mode: prefill + disaggregation-transfer-backend: nixl + tensor-parallel-size: 8 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: total_tokens + chunked-prefill-size: 65536 + max-prefill-tokens: 16384 + max-running-requests: 16 + cuda-graph-max-bs: 16 + disable-cuda-graph: true + nsa-prefill-backend: trtllm + nsa-decode-backend: trtllm + moe-runner-backend: flashinfer_cutlass + fp4-gemm-backend: flashinfer_cutlass + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.8 + context-length: 1048576 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-size: 135 + hicache-io-backend: direct + speculative-algorithm: EAGLE + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 2 + enable-metrics: true + enable-cache-report: true + decode: + nodes: 4 + workers: 4 + gpus: 4 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_TE_METRIC: 'true' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_ENABLE_THINKING: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + SGLANG_REASONING_EFFORT: max + PIP_BREAK_SYSTEM_PACKAGES: '1' + UCX_TLS: cuda_copy,cuda_ipc,sm,self,tcp + NVSHMEM_REMOTE_TRANSPORT: none + SGLANG_DG_CACHE_DIR: /deepgemm_cache + FLASHINFER_WORKSPACE_BASE: /flashinfer_cache + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + disaggregation-mode: decode + disaggregation-transfer-backend: nixl + disable-radix-cache: true + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + max-running-requests: 16 + cuda-graph-max-bs: 16 + chunked-prefill-size: 64 + context-length: 1048576 + nsa-prefill-backend: trtllm + nsa-decode-backend: trtllm + moe-runner-backend: flashinfer_trtllm + fp4-gemm-backend: flashinfer_cutlass + skip-tokenizer-init: true + stream-interval: 30 + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.9 + disaggregation-decode-extra-slots: 0 + enable-metrics: true + enable-cache-report: true + +override_disagg_1p6d_dep8_tp4_c45_mtp: + name: disagg-gb200-1p6d-dep8-tp4-c45-mtp + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_TE_METRIC: 'true' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_ENABLE_THINKING: '1' + SGLANG_ENABLE_UNIFIED_RADIX_TREE: '1' + SGLANG_HICACHE_DEBUG_LOG: '1' + SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' + SGLANG_REASONING_EFFORT: max + PIP_BREAK_SYSTEM_PACKAGES: '1' + UCX_TLS: cuda_copy,cuda_ipc,sm,self,tcp + NVSHMEM_REMOTE_TRANSPORT: none + SGLANG_DG_CACHE_DIR: /deepgemm_cache + FLASHINFER_WORKSPACE_BASE: /flashinfer_cache + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + disaggregation-mode: prefill + disaggregation-transfer-backend: nixl + tensor-parallel-size: 8 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: total_tokens + chunked-prefill-size: 65536 + max-prefill-tokens: 16384 + max-running-requests: 16 + cuda-graph-max-bs: 16 + disable-cuda-graph: true + nsa-prefill-backend: trtllm + nsa-decode-backend: trtllm + moe-runner-backend: flashinfer_cutlass + fp4-gemm-backend: flashinfer_cutlass + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.8 + context-length: 1048576 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-size: 135 + hicache-io-backend: direct + speculative-algorithm: EAGLE + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 2 + enable-metrics: true + enable-cache-report: true + decode: + nodes: 6 + workers: 6 + gpus: 4 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_TE_METRIC: 'true' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_ENABLE_THINKING: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + SGLANG_REASONING_EFFORT: max + PIP_BREAK_SYSTEM_PACKAGES: '1' + UCX_TLS: cuda_copy,cuda_ipc,sm,self,tcp + NVSHMEM_REMOTE_TRANSPORT: none + SGLANG_DG_CACHE_DIR: /deepgemm_cache + FLASHINFER_WORKSPACE_BASE: /flashinfer_cache + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + disaggregation-mode: decode + disaggregation-transfer-backend: nixl + disable-radix-cache: true + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + max-running-requests: 16 + cuda-graph-max-bs: 16 + chunked-prefill-size: 64 + context-length: 1048576 + nsa-prefill-backend: trtllm + nsa-decode-backend: trtllm + moe-runner-backend: flashinfer_trtllm + fp4-gemm-backend: flashinfer_cutlass + skip-tokenizer-init: true + stream-interval: 30 + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.9 + disaggregation-decode-extra-slots: 0 + enable-metrics: true + enable-cache-report: true + +override_disagg_2p1d_dep8_dep16_c128_mtp: + name: disagg-gb200-2p1d-dep8-dep16-c128-mtp + roles: + prefill: + nodes: 4 + workers: 2 + gpus: 8 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_TE_METRIC: 'true' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_ENABLE_THINKING: '1' + SGLANG_ENABLE_UNIFIED_RADIX_TREE: '1' + SGLANG_HICACHE_DEBUG_LOG: '1' + SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' + SGLANG_REASONING_EFFORT: max + PIP_BREAK_SYSTEM_PACKAGES: '1' + UCX_TLS: cuda_copy,cuda_ipc,sm,self,tcp + NVSHMEM_REMOTE_TRANSPORT: none + SGLANG_DG_CACHE_DIR: /deepgemm_cache + FLASHINFER_WORKSPACE_BASE: /flashinfer_cache + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + disaggregation-mode: prefill + disaggregation-transfer-backend: nixl + tensor-parallel-size: 8 + data-parallel-size: 8 + expert-parallel-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: total_tokens + chunked-prefill-size: 65536 + max-prefill-tokens: 8192 + max-running-requests: 16 + cuda-graph-max-bs: 16 + disable-cuda-graph: true + nsa-prefill-backend: trtllm + nsa-decode-backend: trtllm + moe-runner-backend: flashinfer_cutlass + fp4-gemm-backend: flashinfer_cutlass + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.8 + context-length: 1048576 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-size: 100 + hicache-io-backend: direct + speculative-algorithm: EAGLE + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 2 + enable-metrics: true + enable-cache-report: true + decode: + nodes: 4 + workers: 1 + gpus: 16 + env: + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + PYTHONUNBUFFERED: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + MC_TE_METRIC: 'true' + MC_FORCE_MNNVL: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_ENABLE_THINKING: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + SGLANG_REASONING_EFFORT: max + PIP_BREAK_SYSTEM_PACKAGES: '1' + UCX_TLS: cuda_copy,cuda_ipc,sm,self,tcp + NVSHMEM_REMOTE_TRANSPORT: none + SGLANG_DG_CACHE_DIR: /deepgemm_cache + FLASHINFER_WORKSPACE_BASE: /flashinfer_cache + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + disaggregation-mode: decode + disaggregation-transfer-backend: nixl + disable-radix-cache: true + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + tensor-parallel-size: 16 + data-parallel-size: 16 + expert-parallel-size: 16 + enable-dp-attention: true + enable-dp-lm-head: true + max-running-requests: 144 + cuda-graph-max-bs: 144 + chunked-prefill-size: 64 + context-length: 1048576 + nsa-prefill-backend: trtllm + nsa-decode-backend: trtllm + moe-runner-backend: flashinfer_cutedsl + fp4-gemm-backend: flashinfer_cutlass + skip-tokenizer-init: true + stream-interval: 30 + enable-flashinfer-allreduce-fusion: true + weight-loader-prefetch-checkpoints: true + model-loader-extra-config: '{"enable_multithread_load": true}' + mem-fraction-static: 0.85 + disaggregation-decode-extra-slots: 0 + enable-metrics: true + enable-cache-report: true + deepep-config: /configs/deepep_config.json + deepep-mode: low_latency + ep-dispatch-algorithm: static + ep-num-redundant-experts: 0 + moe-a2a-backend: deepep + moe-dense-tp-size: 1 + speculative-moe-a2a-backend: deepep + speculative-moe-runner-backend: deep_gemm diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p1d-tep8-c20-b5-mtp5.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p1d-tep8-c20-b5-mtp5.yaml deleted file mode 100644 index 2d0380aec2..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p1d-tep8-c20-b5-mtp5.yaml +++ /dev/null @@ -1,188 +0,0 @@ -schema: 2 -name: dynamo-disagg-gb300-1p1d-tep8-c20-b5-mtp5 - -model: - path: nvidia/GLM-5.2-NVFP4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc26.dev202609040000 - precision: fp4 - -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - request_plane: tcp - -health_check: - max_attempts: 270 - interval_seconds: 10 - -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false - -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: &server_environment - HF_HUB_OFFLINE: "1" - TRANSFORMERS_OFFLINE: "1" - TQDM_DISABLE: "1" - HF_HUB_DISABLE_PROGRESS_BARS: "1" - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - MIMALLOC_PURGE_DELAY: "0" - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TRTLLM_FUSED_DSA_METADATA: "1" - TRTLLM_DSA_INDEXER_BF16: "1" - TRTLLM_SERVE_ENABLE_MSGSPEC: "1" - TLLM_ADP_ROUTER_MATCH_RATE_THRESHOLD: "0.10" - TRTLLM_KV_CACHE_TRANSFER_TIMEOUT_SEC: "600" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_RNDV_SCHEME: put_zcopy - TRTLLM_KVCACHE_SEND_BUFFER_COUNT: "1" - TRTLLM_KVCACHE_RECV_BUFFER_COUNT: "1" - DYN_TRTLLM_ENABLE_ATTENTION_DP: "1" - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TOKENIZER: fastokens - DYN_PUBLISH_KV_EVENTS: "0" - args: - attention_dp_config: - enable_kv_cache_aware_routing: false - kv_cache_routing_conversation_affinity: true - kv_cache_routing_max_sessions: 65536 - cache_transceiver_config: - backend: NIXL - transceiver_runtime: PYTHON - kv_cache_bounce_size_mb: 5120 - max_tokens_in_buffer: 1048576 - kv_transfer_timeout_ms: 600000 - cuda_graph_config: - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - trust_remote_code: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.75 - host_cache_size: 137438953472 - tokens_per_block: 64 - max_batch_size: 256 - max_num_tokens: 8192 - max_seq_len: 1048576 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - num_postprocess_workers: 8 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - sparse_attention_config: - algorithm: dsa - enable_heuristic_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 5 - tensor_parallel_size: 4 - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: *server_environment - args: - cache_transceiver_config: - backend: NIXL - transceiver_runtime: PYTHON - kv_cache_bounce_size_mb: 5120 - max_tokens_in_buffer: 1048576 - kv_transfer_timeout_ms: 600000 - cuda_graph_config: - batch_sizes: [1, 2, 4, 5] - enable_padding: true - enable_attention_dp: false - trust_remote_code: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - tokens_per_block: 64 - max_batch_size: 5 - max_num_tokens: 128 - max_seq_len: 1048576 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - sparse_attention_config: - algorithm: dsa - enable_heuristic_topk: true - use_cute_dsl_paged_mqa_logits: true - use_cute_dsl_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 5 - stream_interval: 20 - tensor_parallel_size: 8 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_ROUTER_QUEUE_THRESHOLD: None - DYN_ROUTER_TEMPERATURE: '0' - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TOKENIZER: fastokens - DYN_TCP_REQUEST_TIMEOUT: '30' - DYN_LOG: warn - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - placement: - node: head -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: "0" - SERVED_MODEL_NAME: GLM-5.2-NVFP4 - MAX_MODEL_LEN: '1048576' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - OPENAI_API_KEY: EMPTY - KV_OFFLOADING: dram - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p1d-tp8-c1-b1-mtp5.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p1d-tp8-c1-b1-mtp5.yaml deleted file mode 100644 index dcf151af95..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p1d-tp8-c1-b1-mtp5.yaml +++ /dev/null @@ -1,185 +0,0 @@ -schema: 2 -name: dynamo-disagg-gb300-1p1d-tp8-c1-b1-mtp5 - -model: - path: nvidia/GLM-5.2-NVFP4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc26.dev202609040000 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - request_plane: tcp - -health_check: - max_attempts: 270 - interval_seconds: 10 - -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false - -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: &server_environment - HF_HUB_OFFLINE: "1" - TRANSFORMERS_OFFLINE: "1" - TQDM_DISABLE: "1" - HF_HUB_DISABLE_PROGRESS_BARS: "1" - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - MIMALLOC_PURGE_DELAY: "0" - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TRTLLM_FUSED_DSA_METADATA: "1" - TRTLLM_DSA_INDEXER_BF16: "1" - TRTLLM_SERVE_ENABLE_MSGSPEC: "1" - TLLM_ADP_ROUTER_MATCH_RATE_THRESHOLD: "0.10" - TRTLLM_KV_CACHE_TRANSFER_TIMEOUT_SEC: "600" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_RNDV_SCHEME: put_zcopy - TRTLLM_KVCACHE_SEND_BUFFER_COUNT: "1" - TRTLLM_KVCACHE_RECV_BUFFER_COUNT: "1" - DYN_TRTLLM_ENABLE_ATTENTION_DP: "1" - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TOKENIZER: fastokens - DYN_PUBLISH_KV_EVENTS: "0" - args: - attention_dp_config: - enable_kv_cache_aware_routing: false - kv_cache_routing_conversation_affinity: true - kv_cache_routing_max_sessions: 65536 - cache_transceiver_config: - backend: NIXL - transceiver_runtime: PYTHON - kv_cache_bounce_size_mb: 5120 - max_tokens_in_buffer: 1048576 - kv_transfer_timeout_ms: 600000 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - enable_chunked_prefill: true - trust_remote_code: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.75 - host_cache_size: 137438953472 - tokens_per_block: 64 - max_batch_size: 256 - max_num_tokens: 8192 - max_seq_len: 1048576 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - num_postprocess_workers: 8 - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - context_chunking_policy: EQUAL_PROGRESS - sparse_attention_config: - algorithm: dsa - enable_heuristic_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 5 - tensor_parallel_size: 4 - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: *server_environment - args: - cache_transceiver_config: - backend: NIXL - transceiver_runtime: PYTHON - kv_cache_bounce_size_mb: 5120 - max_tokens_in_buffer: 1048576 - kv_transfer_timeout_ms: 600000 - cuda_graph_config: - batch_sizes: [1] - enable_padding: true - enable_attention_dp: false - trust_remote_code: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - tokens_per_block: 64 - max_batch_size: 1 - max_num_tokens: 128 - max_seq_len: 1048576 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 1 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - sparse_attention_config: - algorithm: dsa - enable_heuristic_topk: true - use_cute_dsl_paged_mqa_logits: true - use_cute_dsl_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 5 - stream_interval: 20 - tensor_parallel_size: 8 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_ROUTER_QUEUE_THRESHOLD: None - DYN_ROUTER_TEMPERATURE: '0' - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TOKENIZER: fastokens - DYN_TCP_REQUEST_TIMEOUT: '30' - DYN_LOG: warn - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - placement: - node: head -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: "0" - SERVED_MODEL_NAME: GLM-5.2-NVFP4 - MAX_MODEL_LEN: '1048576' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - OPENAI_API_KEY: EMPTY - KV_OFFLOADING: dram - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p4d-tep4-c30-b2-mtp5.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p4d-tep4-c30-b2-mtp5.yaml deleted file mode 100644 index 9408847d11..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p4d-tep4-c30-b2-mtp5.yaml +++ /dev/null @@ -1,188 +0,0 @@ -schema: 2 -name: dynamo-disagg-gb300-1p4d-tep4-c30-b2-mtp5 - -model: - path: nvidia/GLM-5.2-NVFP4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc26.dev202609040000 - precision: fp4 - -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - request_plane: tcp - -health_check: - max_attempts: 270 - interval_seconds: 10 - -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false - -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: &server_environment - HF_HUB_OFFLINE: "1" - TRANSFORMERS_OFFLINE: "1" - TQDM_DISABLE: "1" - HF_HUB_DISABLE_PROGRESS_BARS: "1" - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - MIMALLOC_PURGE_DELAY: "0" - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TRTLLM_FUSED_DSA_METADATA: "1" - TRTLLM_DSA_INDEXER_BF16: "1" - TRTLLM_SERVE_ENABLE_MSGSPEC: "1" - TLLM_ADP_ROUTER_MATCH_RATE_THRESHOLD: "0.10" - TRTLLM_KV_CACHE_TRANSFER_TIMEOUT_SEC: "600" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_RNDV_SCHEME: put_zcopy - TRTLLM_KVCACHE_SEND_BUFFER_COUNT: "1" - TRTLLM_KVCACHE_RECV_BUFFER_COUNT: "1" - DYN_TRTLLM_ENABLE_ATTENTION_DP: "1" - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TOKENIZER: fastokens - DYN_PUBLISH_KV_EVENTS: "0" - args: - attention_dp_config: - enable_kv_cache_aware_routing: false - kv_cache_routing_conversation_affinity: true - kv_cache_routing_max_sessions: 65536 - cache_transceiver_config: - backend: NIXL - transceiver_runtime: PYTHON - kv_cache_bounce_size_mb: 5120 - max_tokens_in_buffer: 1048576 - kv_transfer_timeout_ms: 600000 - cuda_graph_config: - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - trust_remote_code: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.75 - host_cache_size: 137438953472 - tokens_per_block: 64 - max_batch_size: 256 - max_num_tokens: 8192 - max_seq_len: 1048576 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - num_postprocess_workers: 8 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - sparse_attention_config: - algorithm: dsa - enable_heuristic_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 5 - tensor_parallel_size: 4 - decode: - nodes: 4 - workers: 4 - gpus: 4 - - env: *server_environment - args: - cache_transceiver_config: - backend: NIXL - transceiver_runtime: PYTHON - kv_cache_bounce_size_mb: 5120 - max_tokens_in_buffer: 1048576 - kv_transfer_timeout_ms: 600000 - cuda_graph_config: - batch_sizes: [1, 2] - enable_padding: true - enable_attention_dp: false - trust_remote_code: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - tokens_per_block: 64 - max_batch_size: 2 - max_num_tokens: 128 - max_seq_len: 1048576 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - sparse_attention_config: - algorithm: dsa - enable_heuristic_topk: true - use_cute_dsl_paged_mqa_logits: true - use_cute_dsl_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 5 - stream_interval: 20 - tensor_parallel_size: 4 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_ROUTER_QUEUE_THRESHOLD: None - DYN_ROUTER_TEMPERATURE: '0' - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TOKENIZER: fastokens - DYN_TCP_REQUEST_TIMEOUT: '30' - DYN_LOG: warn - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - placement: - node: head -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: "0" - SERVED_MODEL_NAME: GLM-5.2-NVFP4 - MAX_MODEL_LEN: '1048576' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - OPENAI_API_KEY: EMPTY - KV_OFFLOADING: dram - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-3p4d-tep4-c60-b5-mtp5.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-3p4d-tep4-c60-b5-mtp5.yaml deleted file mode 100644 index 507a1c08eb..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-3p4d-tep4-c60-b5-mtp5.yaml +++ /dev/null @@ -1,188 +0,0 @@ -schema: 2 -name: dynamo-disagg-gb300-3p4d-tep4-c60-b5-mtp5 - -model: - path: nvidia/GLM-5.2-NVFP4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc26.dev202609040000 - precision: fp4 - -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - request_plane: tcp - -health_check: - max_attempts: 270 - interval_seconds: 10 - -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false - -roles: - prefill: - nodes: 3 - workers: 3 - gpus: 4 - env: &server_environment - HF_HUB_OFFLINE: "1" - TRANSFORMERS_OFFLINE: "1" - TQDM_DISABLE: "1" - HF_HUB_DISABLE_PROGRESS_BARS: "1" - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - MIMALLOC_PURGE_DELAY: "0" - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TRTLLM_FUSED_DSA_METADATA: "1" - TRTLLM_DSA_INDEXER_BF16: "1" - TRTLLM_SERVE_ENABLE_MSGSPEC: "1" - TLLM_ADP_ROUTER_MATCH_RATE_THRESHOLD: "0.10" - TRTLLM_KV_CACHE_TRANSFER_TIMEOUT_SEC: "600" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_RNDV_SCHEME: put_zcopy - TRTLLM_KVCACHE_SEND_BUFFER_COUNT: "1" - TRTLLM_KVCACHE_RECV_BUFFER_COUNT: "1" - DYN_TRTLLM_ENABLE_ATTENTION_DP: "1" - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TOKENIZER: fastokens - DYN_PUBLISH_KV_EVENTS: "0" - args: - attention_dp_config: - enable_kv_cache_aware_routing: false - kv_cache_routing_conversation_affinity: true - kv_cache_routing_max_sessions: 65536 - cache_transceiver_config: - backend: NIXL - transceiver_runtime: PYTHON - kv_cache_bounce_size_mb: 5120 - max_tokens_in_buffer: 1048576 - kv_transfer_timeout_ms: 600000 - cuda_graph_config: - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - trust_remote_code: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.75 - host_cache_size: 137438953472 - tokens_per_block: 64 - max_batch_size: 256 - max_num_tokens: 8192 - max_seq_len: 1048576 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - num_postprocess_workers: 8 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - sparse_attention_config: - algorithm: dsa - enable_heuristic_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 5 - tensor_parallel_size: 4 - decode: - nodes: 4 - workers: 4 - gpus: 4 - - env: *server_environment - args: - cache_transceiver_config: - backend: NIXL - transceiver_runtime: PYTHON - kv_cache_bounce_size_mb: 5120 - max_tokens_in_buffer: 1048576 - kv_transfer_timeout_ms: 600000 - cuda_graph_config: - batch_sizes: [1, 2, 4, 5] - enable_padding: true - enable_attention_dp: false - trust_remote_code: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - tokens_per_block: 64 - max_batch_size: 5 - max_num_tokens: 128 - max_seq_len: 1048576 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - sparse_attention_config: - algorithm: dsa - enable_heuristic_topk: true - use_cute_dsl_paged_mqa_logits: true - use_cute_dsl_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 5 - stream_interval: 20 - tensor_parallel_size: 4 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_ROUTER_QUEUE_THRESHOLD: None - DYN_ROUTER_TEMPERATURE: '0' - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TOKENIZER: fastokens - DYN_TCP_REQUEST_TIMEOUT: '30' - DYN_LOG: warn - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - placement: - node: head -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: "0" - SERVED_MODEL_NAME: GLM-5.2-NVFP4 - MAX_MODEL_LEN: '1048576' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - OPENAI_API_KEY: EMPTY - KV_OFFLOADING: dram - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-4p1d-dep8-c227-b16-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-4p1d-dep8-c227-b16-mtp3.yaml deleted file mode 100644 index bb733a601e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-4p1d-dep8-c227-b16-mtp3.yaml +++ /dev/null @@ -1,189 +0,0 @@ -schema: 2 -name: dynamo-disagg-gb300-4p1d-dep8-c227-b16-mtp3 - -model: - path: nvidia/GLM-5.2-NVFP4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc26.dev202609040000 - precision: fp4 - -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - request_plane: tcp - -health_check: - max_attempts: 270 - interval_seconds: 10 - -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false - -roles: - prefill: - nodes: 4 - workers: 4 - gpus: 4 - env: &server_environment - HF_HUB_OFFLINE: "1" - TRANSFORMERS_OFFLINE: "1" - TQDM_DISABLE: "1" - HF_HUB_DISABLE_PROGRESS_BARS: "1" - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - MIMALLOC_PURGE_DELAY: "0" - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TRTLLM_FUSED_DSA_METADATA: "1" - TRTLLM_DSA_INDEXER_BF16: "1" - TRTLLM_SERVE_ENABLE_MSGSPEC: "1" - TLLM_ADP_ROUTER_MATCH_RATE_THRESHOLD: "0.10" - TRTLLM_KV_CACHE_TRANSFER_TIMEOUT_SEC: "600" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_RNDV_SCHEME: put_zcopy - TRTLLM_KVCACHE_SEND_BUFFER_COUNT: "1" - TRTLLM_KVCACHE_RECV_BUFFER_COUNT: "1" - DYN_TRTLLM_ENABLE_ATTENTION_DP: "1" - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TOKENIZER: fastokens - DYN_PUBLISH_KV_EVENTS: "0" - args: - attention_dp_config: - enable_kv_cache_aware_routing: false - kv_cache_routing_conversation_affinity: true - kv_cache_routing_max_sessions: 65536 - cache_transceiver_config: - backend: NIXL - transceiver_runtime: PYTHON - kv_cache_bounce_size_mb: 5120 - max_tokens_in_buffer: 1048576 - kv_transfer_timeout_ms: 600000 - cuda_graph_config: - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - trust_remote_code: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.75 - host_cache_size: 137438953472 - tokens_per_block: 64 - max_batch_size: 256 - max_num_tokens: 8192 - max_seq_len: 1048576 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - num_postprocess_workers: 8 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - sparse_attention_config: - algorithm: dsa - enable_heuristic_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 4 - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: *server_environment - args: - cache_transceiver_config: - backend: NIXL - transceiver_runtime: PYTHON - kv_cache_bounce_size_mb: 5120 - max_tokens_in_buffer: 1048576 - kv_transfer_timeout_ms: 600000 - cuda_graph_config: - batch_sizes: [1, 2, 4, 8, 16] - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - trust_remote_code: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.9 - host_cache_size: 137438953472 - tokens_per_block: 64 - max_batch_size: 16 - max_num_tokens: 128 - max_seq_len: 1048576 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - sparse_attention_config: - algorithm: dsa - enable_heuristic_topk: true - use_cute_dsl_paged_mqa_logits: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 20 - tensor_parallel_size: 8 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_ROUTER_QUEUE_THRESHOLD: None - DYN_ROUTER_TEMPERATURE: '0' - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TOKENIZER: fastokens - DYN_TCP_REQUEST_TIMEOUT: '30' - DYN_LOG: warn - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - placement: - node: head -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: "0" - SERVED_MODEL_NAME: GLM-5.2-NVFP4 - MAX_MODEL_LEN: '1048576' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - OPENAI_API_KEY: EMPTY - KV_OFFLOADING: dram - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-5p1d-dep16-c260-b16-mtp3.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-5p1d-dep16-c260-b16-mtp3.yaml deleted file mode 100644 index 7a9647373a..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-5p1d-dep16-c260-b16-mtp3.yaml +++ /dev/null @@ -1,189 +0,0 @@ -schema: 2 -name: dynamo-disagg-gb300-5p1d-dep16-c260-b16-mtp3 - -model: - path: nvidia/GLM-5.2-NVFP4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc26.dev202609040000 - precision: fp4 - -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - request_plane: tcp - -health_check: - max_attempts: 270 - interval_seconds: 10 - -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false - -roles: - prefill: - nodes: 5 - workers: 5 - gpus: 4 - env: &server_environment - HF_HUB_OFFLINE: "1" - TRANSFORMERS_OFFLINE: "1" - TQDM_DISABLE: "1" - HF_HUB_DISABLE_PROGRESS_BARS: "1" - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: "1" - TRTLLM_WORKER_DISABLE_GC: "1" - TRTLLM_ENABLE_PDL: "1" - NCCL_GRAPH_MIXING_SUPPORT: "0" - MIMALLOC_PURGE_DELAY: "0" - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TRTLLM_FUSED_DSA_METADATA: "1" - TRTLLM_DSA_INDEXER_BF16: "1" - TRTLLM_SERVE_ENABLE_MSGSPEC: "1" - TLLM_ADP_ROUTER_MATCH_RATE_THRESHOLD: "0.10" - TRTLLM_KV_CACHE_TRANSFER_TIMEOUT_SEC: "600" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_RNDV_SCHEME: put_zcopy - TRTLLM_KVCACHE_SEND_BUFFER_COUNT: "1" - TRTLLM_KVCACHE_RECV_BUFFER_COUNT: "1" - DYN_TRTLLM_ENABLE_ATTENTION_DP: "1" - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TOKENIZER: fastokens - DYN_PUBLISH_KV_EVENTS: "0" - args: - attention_dp_config: - enable_kv_cache_aware_routing: false - kv_cache_routing_conversation_affinity: true - kv_cache_routing_max_sessions: 65536 - cache_transceiver_config: - backend: NIXL - transceiver_runtime: PYTHON - kv_cache_bounce_size_mb: 5120 - max_tokens_in_buffer: 1048576 - kv_transfer_timeout_ms: 600000 - cuda_graph_config: - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - trust_remote_code: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.75 - host_cache_size: 137438953472 - tokens_per_block: 64 - max_batch_size: 256 - max_num_tokens: 8192 - max_seq_len: 1048576 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - num_postprocess_workers: 8 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - sparse_attention_config: - algorithm: dsa - enable_heuristic_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 4 - decode: - nodes: 4 - workers: 1 - gpus: 16 - - env: *server_environment - args: - cache_transceiver_config: - backend: NIXL - transceiver_runtime: PYTHON - kv_cache_bounce_size_mb: 5120 - max_tokens_in_buffer: 1048576 - kv_transfer_timeout_ms: 600000 - cuda_graph_config: - batch_sizes: [1, 2, 4, 8, 16] - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - trust_remote_code: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.9 - tokens_per_block: 64 - max_batch_size: 16 - max_num_tokens: 128 - max_seq_len: 1048576 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - pipeline_parallel_size: 1 - print_iter_log: true - enable_iter_perf_stats: true - return_perf_metrics: false - sparse_attention_config: - algorithm: dsa - enable_heuristic_topk: true - use_cute_dsl_paged_mqa_logits: true - use_cute_dsl_topk: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 20 - tensor_parallel_size: 16 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_ROUTER_QUEUE_THRESHOLD: None - DYN_ROUTER_TEMPERATURE: '0' - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TOKENIZER: fastokens - DYN_TCP_REQUEST_TIMEOUT: '30' - DYN_LOG: warn - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - placement: - node: head -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: "0" - SERVED_MODEL_NAME: GLM-5.2-NVFP4 - MAX_MODEL_LEN: '1048576' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - OPENAI_API_KEY: EMPTY - KV_OFFLOADING: dram - placement: - node: dedicated diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..39f3cab1ea --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml @@ -0,0 +1,421 @@ +# srt-slurm recipes for glm5.2/trtllm/gb300-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: nvidia/GLM-5.2-NVFP4 + container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc26.dev202609040000 + precision: fp4 + dynamo: + install: true + source: {} + request_plane: tcp + health_check: + max_attempts: 270 + interval_seconds: 10 + resources: + gpu_type: gb300 + gpus_per_node: 4 + engine: + type: trtllm + publish_events_and_metrics: false + roles: + prefill: + gpus: 4 + env: + HF_HUB_OFFLINE: '1' + TRANSFORMERS_OFFLINE: '1' + TQDM_DISABLE: '1' + HF_HUB_DISABLE_PROGRESS_BARS: '1' + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + TRTLLM_ENABLE_PDL: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + MIMALLOC_PURGE_DELAY: '0' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + TRTLLM_FUSED_DSA_METADATA: '1' + TRTLLM_DSA_INDEXER_BF16: '1' + TRTLLM_SERVE_ENABLE_MSGSPEC: '1' + TLLM_ADP_ROUTER_MATCH_RATE_THRESHOLD: '0.10' + TRTLLM_KV_CACHE_TRANSFER_TIMEOUT_SEC: '600' + UCX_CUDA_IPC_ENABLE_MNNVL: y + UCX_RNDV_SCHEME: put_zcopy + TRTLLM_KVCACHE_SEND_BUFFER_COUNT: '1' + TRTLLM_KVCACHE_RECV_BUFFER_COUNT: '1' + DYN_TRTLLM_ENABLE_ATTENTION_DP: '1' + DYN_ENGINE_CONV_AFFINITY: '1' + DYN_TOKENIZER: fastokens + DYN_PUBLISH_KV_EVENTS: '0' + args: + attention_dp_config: + enable_kv_cache_aware_routing: false + kv_cache_routing_conversation_affinity: true + kv_cache_routing_max_sessions: 65536 + cache_transceiver_config: + backend: NIXL + transceiver_runtime: PYTHON + kv_cache_bounce_size_mb: 5120 + max_tokens_in_buffer: 1048576 + kv_transfer_timeout_ms: 600000 + cuda_graph_config: null + enable_attention_dp: true + enable_chunked_prefill: true + trust_remote_code: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: true + event_buffer_max_size: 0 + free_gpu_memory_fraction: 0.75 + host_cache_size: 137438953472 + tokens_per_block: 64 + max_batch_size: 256 + max_num_tokens: 8192 + max_seq_len: 1048576 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 4 + num_postprocess_workers: 8 + pipeline_parallel_size: 1 + print_iter_log: true + return_perf_metrics: false + scheduler_config: + capacity_scheduler_policy: MAX_UTILIZATION + sparse_attention_config: + algorithm: dsa + enable_heuristic_topk: true + speculative_config: + decoding_type: MTP + tensor_parallel_size: 4 + decode: + env: + HF_HUB_OFFLINE: '1' + TRANSFORMERS_OFFLINE: '1' + TQDM_DISABLE: '1' + HF_HUB_DISABLE_PROGRESS_BARS: '1' + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + TRTLLM_ENABLE_PDL: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + MIMALLOC_PURGE_DELAY: '0' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + TRTLLM_FUSED_DSA_METADATA: '1' + TRTLLM_DSA_INDEXER_BF16: '1' + TRTLLM_SERVE_ENABLE_MSGSPEC: '1' + TLLM_ADP_ROUTER_MATCH_RATE_THRESHOLD: '0.10' + TRTLLM_KV_CACHE_TRANSFER_TIMEOUT_SEC: '600' + UCX_CUDA_IPC_ENABLE_MNNVL: y + UCX_RNDV_SCHEME: put_zcopy + TRTLLM_KVCACHE_SEND_BUFFER_COUNT: '1' + TRTLLM_KVCACHE_RECV_BUFFER_COUNT: '1' + DYN_TRTLLM_ENABLE_ATTENTION_DP: '1' + DYN_ENGINE_CONV_AFFINITY: '1' + DYN_TOKENIZER: fastokens + DYN_PUBLISH_KV_EVENTS: '0' + args: + cache_transceiver_config: + backend: NIXL + transceiver_runtime: PYTHON + kv_cache_bounce_size_mb: 5120 + max_tokens_in_buffer: 1048576 + kv_transfer_timeout_ms: 600000 + cuda_graph_config: + enable_padding: true + trust_remote_code: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + event_buffer_max_size: 0 + tokens_per_block: 64 + max_num_tokens: 128 + max_seq_len: 1048576 + moe_config: + use_low_precision_moe_combine: true + num_postprocess_workers: 4 + pipeline_parallel_size: 1 + print_iter_log: true + return_perf_metrics: false + sparse_attention_config: + algorithm: dsa + enable_heuristic_topk: true + use_cute_dsl_paged_mqa_logits: true + speculative_config: + decoding_type: MTP + stream_interval: 20 + frontend: + type: dynamo + enable_multiple_frontends: false + env: + ETCD_LEASE_TTL: '120' + DYN_ROUTER_QUEUE_THRESHOLD: None + DYN_ROUTER_TEMPERATURE: '0' + DYN_TOKENIZER_CACHE: '1' + DYN_TOKENIZER_CACHE_BYTES: '8000000000' + DYN_TOKENIZER: fastokens + DYN_TCP_REQUEST_TIMEOUT: '30' + DYN_LOG: warn + DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' + args: + router-mode: kv + no-kv-events: true + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + placement: + node: head + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: '0' + SERVED_MODEL_NAME: GLM-5.2-NVFP4 + MAX_MODEL_LEN: '1048576' + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + OPENAI_API_KEY: EMPTY + KV_OFFLOADING: dram + placement: + node: dedicated + +override_disagg_1p1d_tep8_c20_b5_mtp5: + name: dynamo-disagg-gb300-1p1d-tep8-c20-b5-mtp5 + dynamo: + source: + git: https://github.com/cquil11/dynamo.git + rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 + roles: + prefill: + nodes: 1 + workers: 1 + args: + disable_overlap_scheduler: false + enable_iter_perf_stats: true + speculative_config: + max_draft_len: 5 + decode: + nodes: 2 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 5] + enable_attention_dp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 5 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + enable_iter_perf_stats: true + sparse_attention_config: + use_cute_dsl_topk: true + speculative_config: + max_draft_len: 5 + tensor_parallel_size: 8 + benchmark: + env: + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization + +override_disagg_1p1d_tp8_c1_b1_mtp5: + name: dynamo-disagg-gb300-1p1d-tp8-c1-b1-mtp5 + dynamo: + source: + wheel: 1.4.0.dev20260807 + roles: + prefill: + nodes: 1 + workers: 1 + args: + disable_overlap_scheduler: true + scheduler_config: + context_chunking_policy: EQUAL_PROGRESS + speculative_config: + max_draft_len: 5 + decode: + nodes: 2 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + batch_sizes: [1] + enable_attention_dp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 1 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 1 + sparse_attention_config: + use_cute_dsl_topk: true + speculative_config: + max_draft_len: 5 + tensor_parallel_size: 8 + +override_disagg_1p4d_tep4_c30_b2_mtp5: + name: dynamo-disagg-gb300-1p4d-tep4-c30-b2-mtp5 + dynamo: + source: + git: https://github.com/cquil11/dynamo.git + rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 + roles: + prefill: + nodes: 1 + workers: 1 + args: + disable_overlap_scheduler: false + enable_iter_perf_stats: true + speculative_config: + max_draft_len: 5 + decode: + nodes: 4 + workers: 4 + gpus: 4 + args: + cuda_graph_config: + batch_sizes: [1, 2] + enable_attention_dp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 2 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 4 + enable_iter_perf_stats: true + sparse_attention_config: + use_cute_dsl_topk: true + speculative_config: + max_draft_len: 5 + tensor_parallel_size: 4 + benchmark: + env: + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization + +override_disagg_3p4d_tep4_c60_b5_mtp5: + name: dynamo-disagg-gb300-3p4d-tep4-c60-b5-mtp5 + dynamo: + source: + git: https://github.com/cquil11/dynamo.git + rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 + roles: + prefill: + nodes: 3 + workers: 3 + args: + disable_overlap_scheduler: false + enable_iter_perf_stats: true + speculative_config: + max_draft_len: 5 + decode: + nodes: 4 + workers: 4 + gpus: 4 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 5] + enable_attention_dp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 5 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 4 + enable_iter_perf_stats: true + sparse_attention_config: + use_cute_dsl_topk: true + speculative_config: + max_draft_len: 5 + tensor_parallel_size: 4 + benchmark: + env: + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization + +override_disagg_4p1d_dep8_c227_b16_mtp3: + name: dynamo-disagg-gb300-4p1d-dep8-c227-b16-mtp3 + dynamo: + source: + git: https://github.com/cquil11/dynamo.git + rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 + roles: + prefill: + nodes: 4 + workers: 4 + args: + disable_overlap_scheduler: false + enable_iter_perf_stats: true + speculative_config: + max_draft_len: 3 + decode: + nodes: 2 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + enable_attention_dp: true + kv_cache_config: + free_gpu_memory_fraction: 0.9 + host_cache_size: 137438953472 + max_batch_size: 16 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 8 + enable_iter_perf_stats: true + speculative_config: + max_draft_len: 3 + tensor_parallel_size: 8 + enable_lm_head_tp_in_adp: false + benchmark: + env: + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization + +override_disagg_5p1d_dep16_c260_b16_mtp3: + name: dynamo-disagg-gb300-5p1d-dep16-c260-b16-mtp3 + dynamo: + source: + git: https://github.com/cquil11/dynamo.git + rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 + roles: + prefill: + nodes: 5 + workers: 5 + args: + disable_overlap_scheduler: false + enable_iter_perf_stats: true + speculative_config: + max_draft_len: 3 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + enable_attention_dp: true + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 16 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 16 + enable_iter_perf_stats: true + sparse_attention_config: + use_cute_dsl_topk: true + speculative_config: + max_draft_len: 3 + tensor_parallel_size: 16 + enable_lm_head_tp_in_adp: false + benchmark: + env: + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c1.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c1.yaml deleted file mode 100644 index 532de487ee..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c1.yaml +++ /dev/null @@ -1,142 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-b200-tp8pp2-mooncake-c1-agentic" - -model: - path: "kimik3" - container: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - interval_seconds: 10 - max_attempts: 720 - -resources: - gpu_type: "b200" - gpus_per_node: 8 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 16 - - env: - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_RPC_TIMEOUT: "600000" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - NCCL_CUMEM_ENABLE: "1" - TILELANG_CLEANUP_TEMP_FILES: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1,mlx5_4:1,mlx5_5:1,mlx5_10:1,mlx5_11:1" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - pipeline-parallel-size: 2 - decode-context-parallel-size: 8 - dcp-comm-backend: a2a - trust-remote-code: true - load-format: fastsafetensors - moe-backend: auto - gpu-memory-utilization: 0.92 - no-enable-flashinfer-autotune: true - max-model-len: 1048576 - kv-cache-dtype: fp8 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - attention-backend: TOKENSPEED_MLA - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - max-num-seqs: 2 - max-num-batched-tokens: 8192 - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' - prefix-match-unit: 128 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,64,128,256,512,1024,2048,4096,8192]}' - -services: - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -sbatch_directives: - segment: "1" - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - concurrencies: [1] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c14.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c14.yaml deleted file mode 100644 index 590cfc7f7e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c14.yaml +++ /dev/null @@ -1,142 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-b200-tp8pp2-mooncake-c14-agentic" - -model: - path: "kimik3" - container: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - interval_seconds: 10 - max_attempts: 720 - -resources: - gpu_type: "b200" - gpus_per_node: 8 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 16 - - env: - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_RPC_TIMEOUT: "600000" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - NCCL_CUMEM_ENABLE: "1" - TILELANG_CLEANUP_TEMP_FILES: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1,mlx5_4:1,mlx5_5:1,mlx5_10:1,mlx5_11:1" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - pipeline-parallel-size: 2 - decode-context-parallel-size: 8 - dcp-comm-backend: a2a - trust-remote-code: true - load-format: fastsafetensors - moe-backend: auto - gpu-memory-utilization: 0.92 - no-enable-flashinfer-autotune: true - max-model-len: 1048576 - kv-cache-dtype: fp8 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - attention-backend: TOKENSPEED_MLA - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - max-num-seqs: 28 - max-num-batched-tokens: 8192 - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' - prefix-match-unit: 128 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,24,40,56,64,80,104,128,256,512,1024,2048,4096,8192]}' - -services: - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -sbatch_directives: - segment: "1" - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - concurrencies: [14] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c24.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c24.yaml deleted file mode 100644 index cdcb3fb8d0..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c24.yaml +++ /dev/null @@ -1,142 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-b200-tp8pp2-mooncake-c24-agentic" - -model: - path: "kimik3" - container: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - interval_seconds: 10 - max_attempts: 720 - -resources: - gpu_type: "b200" - gpus_per_node: 8 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 16 - - env: - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_RPC_TIMEOUT: "600000" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - NCCL_CUMEM_ENABLE: "1" - TILELANG_CLEANUP_TEMP_FILES: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1,mlx5_4:1,mlx5_5:1,mlx5_10:1,mlx5_11:1" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - pipeline-parallel-size: 2 - decode-context-parallel-size: 8 - dcp-comm-backend: a2a - trust-remote-code: true - load-format: fastsafetensors - moe-backend: auto - gpu-memory-utilization: 0.92 - no-enable-flashinfer-autotune: true - max-model-len: 1048576 - kv-cache-dtype: fp8 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - attention-backend: TOKENSPEED_MLA - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - max-num-seqs: 48 - max-num-batched-tokens: 8192 - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":4,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' - prefix-match-unit: 128 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,24,40,56,64,80,104,128,256,512,1024,2048,4096,8192]}' - -services: - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -sbatch_directives: - segment: "1" - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - concurrencies: [24] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c4.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c4.yaml deleted file mode 100644 index a45f184bec..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c4.yaml +++ /dev/null @@ -1,142 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-b200-tp8pp2-mooncake-c4-agentic" - -model: - path: "kimik3" - container: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - interval_seconds: 10 - max_attempts: 720 - -resources: - gpu_type: "b200" - gpus_per_node: 8 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 16 - - env: - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_RPC_TIMEOUT: "600000" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - NCCL_CUMEM_ENABLE: "1" - TILELANG_CLEANUP_TEMP_FILES: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1,mlx5_4:1,mlx5_5:1,mlx5_10:1,mlx5_11:1" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - pipeline-parallel-size: 2 - decode-context-parallel-size: 8 - dcp-comm-backend: a2a - trust-remote-code: true - load-format: fastsafetensors - moe-backend: auto - gpu-memory-utilization: 0.92 - no-enable-flashinfer-autotune: true - max-model-len: 1048576 - kv-cache-dtype: fp8 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - attention-backend: TOKENSPEED_MLA - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - max-num-seqs: 8 - max-num-batched-tokens: 8192 - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' - prefix-match-unit: 128 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,24,32,40,48,56,64,128,256,512,1024,2048,4096,8192]}' - -services: - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -sbatch_directives: - segment: "1" - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - concurrencies: [4] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c48.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c48.yaml deleted file mode 100644 index a8d44175ac..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c48.yaml +++ /dev/null @@ -1,142 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-b200-tp8pp2-mooncake-c48-agentic" - -model: - path: "kimik3" - container: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - interval_seconds: 10 - max_attempts: 720 - -resources: - gpu_type: "b200" - gpus_per_node: 8 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 16 - - env: - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_RPC_TIMEOUT: "600000" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - NCCL_CUMEM_ENABLE: "1" - TILELANG_CLEANUP_TEMP_FILES: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1,mlx5_4:1,mlx5_5:1,mlx5_10:1,mlx5_11:1" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - pipeline-parallel-size: 2 - decode-context-parallel-size: 8 - dcp-comm-backend: a2a - trust-remote-code: true - load-format: fastsafetensors - moe-backend: auto - gpu-memory-utilization: 0.92 - no-enable-flashinfer-autotune: true - max-model-len: 1048576 - kv-cache-dtype: fp8 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - attention-backend: TOKENSPEED_MLA - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - max-num-seqs: 96 - max-num-batched-tokens: 8192 - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":4,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' - prefix-match-unit: 128 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,7,18,34,53,64,75,100,128,256,512,1024,2048,4096,8192]}' - -services: - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -sbatch_directives: - segment: "1" - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - concurrencies: [48] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c8.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c8.yaml deleted file mode 100644 index 2bff214dd2..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c8.yaml +++ /dev/null @@ -1,142 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-b200-tp8pp2-mooncake-c8-agentic" - -model: - path: "kimik3" - container: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - interval_seconds: 10 - max_attempts: 720 - -resources: - gpu_type: "b200" - gpus_per_node: 8 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 16 - - env: - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_RPC_TIMEOUT: "600000" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - NCCL_CUMEM_ENABLE: "1" - TILELANG_CLEANUP_TEMP_FILES: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1,mlx5_4:1,mlx5_5:1,mlx5_10:1,mlx5_11:1" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - pipeline-parallel-size: 2 - decode-context-parallel-size: 8 - dcp-comm-backend: a2a - trust-remote-code: true - load-format: fastsafetensors - moe-backend: auto - gpu-memory-utilization: 0.92 - no-enable-flashinfer-autotune: true - max-model-len: 1048576 - kv-cache-dtype: fp8 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - attention-backend: TOKENSPEED_MLA - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - max-num-seqs: 16 - max-num-batched-tokens: 8192 - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' - prefix-match-unit: 128 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,24,40,56,64,80,104,128,256,512,1024,2048,4096,8192]}' - -services: - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -sbatch_directives: - segment: "1" - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - concurrencies: [8] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c96.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c96.yaml deleted file mode 100644 index 28018a5910..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c96.yaml +++ /dev/null @@ -1,141 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-b200-tp8pp2-mooncake-c96-agentic" - -model: - path: "kimik3" - container: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - interval_seconds: 10 - max_attempts: 720 - -resources: - gpu_type: "b200" - gpus_per_node: 8 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 16 - - env: - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_RPC_TIMEOUT: "600000" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - NCCL_CUMEM_ENABLE: "1" - TILELANG_CLEANUP_TEMP_FILES: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1,mlx5_4:1,mlx5_5:1,mlx5_10:1,mlx5_11:1" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - pipeline-parallel-size: 2 - decode-context-parallel-size: 8 - dcp-comm-backend: a2a - trust-remote-code: true - load-format: fastsafetensors - moe-backend: auto - gpu-memory-utilization: 0.92 - no-enable-flashinfer-autotune: true - max-model-len: 1048576 - kv-cache-dtype: fp8 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - attention-backend: TOKENSPEED_MLA - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - max-num-seqs: 192 - max-num-batched-tokens: 8192 - prefix-match-unit: 128 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,7,18,34,53,64,75,100,128,256,512,1024,2048,4096,8192]}' - -services: - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -sbatch_directives: - segment: "1" - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - concurrencies: [96] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..4abd197b4b --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml @@ -0,0 +1,207 @@ +# srt-slurm recipes for kimik3/vllm/b200-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: kimik3 + container: vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad + precision: fp4 + identity: + model: + repo: moonshotai/Kimi-K3 + container: + image: vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad + dynamo: + install: false + slurm: + time_limit: '8:00:00' + health_check: + interval_seconds: 10 + max_attempts: 720 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + engine: + type: vllm + connector: null + roles: + agg: + nodes: 2 + workers: 1 + gpus: 16 + env: + VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1' + MC_GID_INDEX: '3' + MC_STORE_MEMCPY: '1' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_SLICE_SIZE: '1048576' + MC_WORKERS_PER_CTX: '4' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_RPC_TIMEOUT: '600000' + WITH_NVIDIA_PEERMEM: '0' + VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0' + NCCL_CUMEM_ENABLE: '1' + TILELANG_CLEANUP_TEMP_FILES: '1' + UCX_MEMTYPE_CACHE: n + UCX_MEMTYPE_REG_WHOLE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1,mlx5_4:1,mlx5_5:1,mlx5_10:1,mlx5_11:1 + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + args: + served-model-name: moonshotai/Kimi-K3 + tensor-parallel-size: 8 + pipeline-parallel-size: 2 + decode-context-parallel-size: 8 + dcp-comm-backend: a2a + trust-remote-code: true + load-format: fastsafetensors + moe-backend: auto + gpu-memory-utilization: 0.92 + no-enable-flashinfer-autotune: true + max-model-len: 1048576 + kv-cache-dtype: fp8 + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' + attention-backend: TOKENSPEED_MLA + attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' + enable-prefix-caching: true + enable-prompt-tokens-details: true + enable-auto-tool-choice: true + tool-call-parser: kimi_k3 + reasoning-parser: kimi_k3 + language-model-only: true + max-num-batched-tokens: 8192 + prefix-match-unit: 128 + services: + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 200GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + sbatch_directives: + segment: '1' + srun_options: + container-remap-root: '' + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + collector_join_timeout_seconds: 12 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: '300' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + +override_agg_tp8pp2_mooncake_c1: + name: kimik3-vllm-agg-b200-tp8pp2-mooncake-c1-agentic + roles: + agg: + args: + max-num-seqs: 2 + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,64,128,256,512,1024,2048,4096,8192]}' + benchmark: + concurrencies: [1] + +override_agg_tp8pp2_mooncake_c14: + name: kimik3-vllm-agg-b200-tp8pp2-mooncake-c14-agentic + roles: + agg: + args: + max-num-seqs: 28 + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,24,40,56,64,80,104,128,256,512,1024,2048,4096,8192]}' + benchmark: + concurrencies: [14] + +override_agg_tp8pp2_mooncake_c24: + name: kimik3-vllm-agg-b200-tp8pp2-mooncake-c24-agentic + roles: + agg: + args: + max-num-seqs: 48 + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":4,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,24,40,56,64,80,104,128,256,512,1024,2048,4096,8192]}' + benchmark: + concurrencies: [24] + +override_agg_tp8pp2_mooncake_c4: + name: kimik3-vllm-agg-b200-tp8pp2-mooncake-c4-agentic + roles: + agg: + args: + max-num-seqs: 8 + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,24,32,40,48,56,64,128,256,512,1024,2048,4096,8192]}' + benchmark: + concurrencies: [4] + +override_agg_tp8pp2_mooncake_c48: + name: kimik3-vllm-agg-b200-tp8pp2-mooncake-c48-agentic + roles: + agg: + args: + max-num-seqs: 96 + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":4,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,7,18,34,53,64,75,100,128,256,512,1024,2048,4096,8192]}' + benchmark: + concurrencies: [48] + +override_agg_tp8pp2_mooncake_c8: + name: kimik3-vllm-agg-b200-tp8pp2-mooncake-c8-agentic + roles: + agg: + args: + max-num-seqs: 16 + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,24,40,56,64,80,104,128,256,512,1024,2048,4096,8192]}' + benchmark: + concurrencies: [8] + +override_agg_tp8pp2_mooncake_c96: + name: kimik3-vllm-agg-b200-tp8pp2-mooncake-c96-agentic + roles: + agg: + args: + max-num-seqs: 192 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,7,18,34,53,64,75,100,128,256,512,1024,2048,4096,8192]}' + benchmark: + concurrencies: [96] diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-dcp16-dspark4-maxseq2-mooncake.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-dcp16-dspark4-maxseq2-mooncake.yaml deleted file mode 100644 index ceb0efb874..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-dcp16-dspark4-maxseq2-mooncake.yaml +++ /dev/null @@ -1,177 +0,0 @@ -# GB200 TP16/DCP16 aggregate profile with DSpark K=4 and max-num-seqs 2. -schema: 2 -name: "kimi-k3-vllm-agg-gb200-dcp16-dspark4-maxseq2-mooncake-agentic" - -model: - path: "kimi-k3" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-75c2eef" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-75c2eef" - frameworks: - dynamo: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" - -dynamo: - install: true - - source: - rev: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" -slurm: - time_limit: "04:00:00" - -health_check: - max_attempts: 720 - interval_seconds: 10 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - args: - - "--eviction_high_watermark_ratio=0.95" - - "--eviction_ratio=0.10" - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "96GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "kimi_k3" - reasoning-parser: "kimi_k3" - enable-auto-tool-choice: true - router-mode: "random" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 16 - - env: - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - # Preserve the measured collective fallback when moving to the built image. - VLLM_USE_DIRECT_DCP_A2A: "0" - VLLM_USE_DIRECT_DCP_Q_GATHER: "0" - VLLM_USE_DIRECT_DCP_KV_GATHER: "0" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - PYTHONNOUSERSITE: "1" - TORCH_CUDA_ARCH_LIST: "10.0" - PYTHONHASHSEED: "42" - VLLM_HTTP_TIMEOUT_KEEP_ALIVE: "900" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 16 - decode-context-parallel-size: 16 - dcp-comm-backend: "a2a" - max-num-seqs: 2 - max-num-batched-tokens: 8192 - trust-remote-code: true - language-model-only: true - dyn-tool-call-parser: "kimi_k3" - reasoning-parser: "kimi_k3" - dyn-reasoning-parser: "kimi_k3" - load-format: "safetensors" - safetensors-load-strategy: "lazy" - moe-backend: "auto" - no-enable-flashinfer-autotune: true - enable-cumem-allocator: true - enable-prefix-caching: true - prefix-match-unit: 128 - kv-cache-dtype: "fp8" - stream-interval: 10 - attention-backend: "TOKENSPEED_MLA" - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - # Throughput jobs inject the committed K=4 golden AL (3.36); EVAL_ONLY - # preserves this real target-verification configuration. - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - max-cudagraph-capture-size: 1024 - kv-cache-memory: 10737418240 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - -sbatch_directives: - mem: "0" - cpus-per-task: "144" - comment: >- - '{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will - take extra loading time."}}' - -srun_options: - mem: "0" - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - client_placement: head - concurrencies: [1, 2, 4, 8, 16] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - ENABLE_AGENTX_POWER: "1" - REQUIRE_POWER: "1" - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-dcp16-nospec-mooncake.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-dcp16-nospec-mooncake.yaml deleted file mode 100644 index 81af8a3005..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-dcp16-nospec-mooncake.yaml +++ /dev/null @@ -1,173 +0,0 @@ -# GB200 TP16/DCP16 aggregate profile without speculative decoding. -schema: 2 -name: "kimi-k3-vllm-agg-gb200-dcp16-nospec-mooncake-agentic" - -model: - path: "kimi-k3" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-75c2eef" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-75c2eef" - frameworks: - dynamo: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" - -dynamo: - install: true - - source: - rev: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" -slurm: - time_limit: "04:00:00" - -health_check: - max_attempts: 720 - interval_seconds: 10 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - args: - - "--eviction_high_watermark_ratio=0.95" - - "--eviction_ratio=0.10" - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "96GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "kimi_k3" - reasoning-parser: "kimi_k3" - enable-auto-tool-choice: true - router-mode: "random" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 16 - - env: - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - # Preserve the measured collective fallback when moving to the built image. - VLLM_USE_DIRECT_DCP_A2A: "0" - VLLM_USE_DIRECT_DCP_Q_GATHER: "0" - VLLM_USE_DIRECT_DCP_KV_GATHER: "0" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - PYTHONNOUSERSITE: "1" - TORCH_CUDA_ARCH_LIST: "10.0" - PYTHONHASHSEED: "42" - VLLM_HTTP_TIMEOUT_KEEP_ALIVE: "900" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 16 - decode-context-parallel-size: 16 - dcp-comm-backend: "a2a" - max-num-batched-tokens: 16384 - trust-remote-code: true - language-model-only: true - dyn-tool-call-parser: "kimi_k3" - reasoning-parser: "kimi_k3" - dyn-reasoning-parser: "kimi_k3" - load-format: "safetensors" - safetensors-load-strategy: "lazy" - moe-backend: "auto" - no-enable-flashinfer-autotune: true - enable-prefix-caching: true - prefix-match-unit: 128 - kv-cache-dtype: "fp8" - stream-interval: 10 - attention-backend: "FLASHINFER_MLA" - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - max-cudagraph-capture-size: 1024 - max-num-seqs: 1000 - kv-cache-memory: 10737418240 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - -sbatch_directives: - mem: "0" - cpus-per-task: "144" - comment: >- - '{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will - take extra loading time."}}' - -srun_options: - mem: "0" - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - client_placement: head - concurrencies: [8, 40, 48] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - ENABLE_AGENTX_POWER: "1" - REQUIRE_POWER: "1" - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-dep16-vllm-simple-offload.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-dep16-vllm-simple-offload.yaml deleted file mode 100644 index ff19c7103b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-dep16-vllm-simple-offload.yaml +++ /dev/null @@ -1,163 +0,0 @@ -schema: 2 -name: "kimi-k3-vllm-agg-gb200-dep16-throughput-vllm-simple-offload-agentic" - -# High-concurrency host-DRAM KV-offload variant of the official throughput- -# oriented multi_node_dep profile. TP4 x DP4 gives EP16 across four four-GPU -# GB200 nodes. Each TP rank receives a 128 GiB CPU KV pool (512 GiB per node). -# https://recipes.vllm.ai/moonshotai/Kimi-K3?hardware=gb200&nodes=4&strategy=multi_node_dep - -model: - path: "kimi-k3" - container: "vllm/vllm-openai:kimi-k3" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:kimi-k3" - frameworks: - dynamo: "1.3.0" - -dynamo: - install: true - - source: - rev: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" -setup_script: kimik3-dspark-config-compat.sh - -environment: - ETCD_LEASE_TTL: "7200" - -slurm: - time_limit: "12:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - args: - # Use vLLM's Kimi K3 parser so OpenAI responses expose structured - # tool_calls instead of raw XTML in message.content. - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "kimi_k3" - reasoning-parser: "kimi_k3" - enable-auto-tool-choice: true - router-mode: "kv" - router-kv-events: true - router-temperature: "0" - router-min-initial-workers: 1 - kv-cache-block-size: 64 - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 16 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - NCCL_P2P_LEVEL: "NVL" - NVIDIA_GDRCOPY: "1" - PYTHONHASHSEED: "42" - PYTORCH_ALLOC_CONF: "expandable_segments:True" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-kimi-k3-gb200-dep16-offload-{job_id}" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 4 - pipeline-parallel-size: 1 - data-parallel-size: 4 - data-parallel-rpc-port: 13345 - enable-expert-parallel: true - trust-remote-code: true - # FlashInfer's larger TP4 MoE representation leaves too little transient - # HBM for fastsafetensors' GPU-staging path when DSpark is loaded. - load-format: "safetensors" - safetensors-load-strategy: "lazy" - kv-cache-dtype: "fp8" - attention-backend: "FLASHINFER_MLA" - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - # DeepGEMM mega-MoE grid barriers time out under DSpark MRV2 with the - # one-sided all-to-all worker; use the supported Kimi K3 FlashInfer path. - moe-backend: "flashinfer_trtllm" - kda-prefill-backend: "flashkda" - kernel-config: '{"enable_cutedsl_warmup":true}' - all2all-backend: "flashinfer_nvlink_one_sided" - gpu-memory-utilization: 0.94 - # The largest offload point is c384 / DP4 = 96 sequences per engine. - # Capture even sequence counts: all configured DP4 steady-state batch - # sizes are exact hits, while odd loads pad by at most one sequence. - max-num-seqs: 96 - max-num-batched-tokens: 16384 - speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - compilation-config: '{"cudagraph_mode":"PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,42,48,54,60,66,72,78,84,90,96,102,108,114,120,126,132,138,144,150,156,162,168,174,180,186,192,198,204,210,216,222,228,234,240,246,252,258,264,270,276,282,288]}' - block-size: 64 - language-model-only: true - disable-custom-all-reduce: true - enable-prefix-caching: true - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":549755813888,"cpu_bytes_to_use_per_rank":137438953472,"lazy_offload":false}}' - scheduler-cls: "vllm.v1.core.sched.async_scheduler.AsyncScheduler" - dyn-tool-call-parser: "kimi_k3" - reasoning-parser: "kimi_k3" - dyn-reasoning-parser: "kimi_k3" - no-enable-flashinfer-autotune: true - -sbatch_directives: - cpus-per-task: "144" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AGENTIC_WARMUP_GRACE_PERIOD: "3600" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-dep16.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-dep16.yaml deleted file mode 100644 index 4edaabdfc5..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-dep16.yaml +++ /dev/null @@ -1,160 +0,0 @@ -schema: 2 -name: "kimi-k3-vllm-agg-gb200-dep16-throughput-agentic" - -# Day-0 GB200 translation of the official throughput-oriented multi_node_dep -# profile. TP4 x DP4 gives EP16 across four four-GPU GB200 nodes. -# https://recipes.vllm.ai/moonshotai/Kimi-K3?hardware=gb200&nodes=4&strategy=multi_node_dep - -model: - path: "kimi-k3" - container: "vllm/vllm-openai:kimi-k3" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:kimi-k3" - frameworks: - dynamo: "1.3.0" - -dynamo: - install: true - - source: - rev: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" -setup_script: kimik3-dspark-config-compat.sh - -environment: - ETCD_LEASE_TTL: "7200" - -slurm: - time_limit: "12:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - args: - # Use vLLM's Kimi K3 parser so OpenAI responses expose structured - # tool_calls instead of raw XTML in message.content. - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "kimi_k3" - reasoning-parser: "kimi_k3" - enable-auto-tool-choice: true - router-mode: "kv" - router-kv-events: true - router-temperature: "0" - router-min-initial-workers: 1 - kv-cache-block-size: 64 - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 16 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - NCCL_P2P_LEVEL: "NVL" - NVIDIA_GDRCOPY: "1" - PYTORCH_ALLOC_CONF: "expandable_segments:True" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-kimi-k3-gb200-dep16-{job_id}" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 4 - pipeline-parallel-size: 1 - data-parallel-size: 4 - data-parallel-rpc-port: 13345 - enable-expert-parallel: true - trust-remote-code: true - # FlashInfer's larger TP4 MoE representation leaves too little transient - # HBM for fastsafetensors' GPU-staging path when DSpark is loaded. - load-format: "safetensors" - safetensors-load-strategy: "lazy" - kv-cache-dtype: "fp8" - attention-backend: "FLASHINFER_MLA" - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - # DeepGEMM mega-MoE grid barriers time out under DSpark MRV2 with the - # one-sided all-to-all worker; use the supported Kimi K3 FlashInfer path. - moe-backend: "flashinfer_trtllm" - kda-prefill-backend: "flashkda" - kernel-config: '{"enable_cutedsl_warmup":true}' - all2all-backend: "flashinfer_nvlink_one_sided" - gpu-memory-utilization: 0.94 - # The largest regular DEP point is c256 / DP4 = 64 sequences per engine. - # Capturing 128 sequence slots consumes 10.9 GiB and leaves too little - # runtime workspace for FlashInfer's MXFP4 MoE kernel. - max-num-seqs: 64 - max-num-batched-tokens: 16384 - speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - compilation-config: '{"cudagraph_mode":"PIECEWISE","cudagraph_capture_sizes":[3,6,9,12,15,18,21,24,27,30,33,36,39,42,45,48,51,54,57,60,63,66,69,72,75,78,81,84,87,90,93,96,99,102,105,108,111,114,117,120,123,126,129,132,135,138,141,144,147,150,153,156,159,162,165,168,171,174,177,180,183,186,189,192]}' - block-size: 64 - language-model-only: true - disable-custom-all-reduce: true - enable-prefix-caching: true - scheduler-cls: "vllm.v1.core.sched.async_scheduler.AsyncScheduler" - dyn-tool-call-parser: "kimi_k3" - reasoning-parser: "kimi_k3" - dyn-reasoning-parser: "kimi_k3" - no-enable-flashinfer-autotune: true - -sbatch_directives: - cpus-per-task: "144" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AGENTIC_WARMUP_GRACE_PERIOD: "3600" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tep16-balanced.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tep16-balanced.yaml deleted file mode 100644 index aace6df184..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tep16-balanced.yaml +++ /dev/null @@ -1,151 +0,0 @@ -schema: 2 -name: "kimi-k3-vllm-agg-gb200-tep16-balanced-agentic" - -# Day-0 GB200 translation of the official balanced multi_node_tep profile. -# Dense layers and MoE experts are sharded across 16 GPUs on four GB200 nodes -# with the official FP8 KV cache. -# https://recipes.vllm.ai/moonshotai/Kimi-K3?hardware=gb200&nodes=4&strategy=multi_node_tep - -model: - path: "kimi-k3" - container: "vllm/vllm-openai:kimi-k3" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:kimi-k3" - frameworks: - dynamo: "1.3.0" - -dynamo: - install: true - - source: - rev: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" -setup_script: kimik3-dspark-config-compat.sh - -environment: - ETCD_LEASE_TTL: "7200" - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - args: - # Use vLLM's Kimi K3 parser so OpenAI responses expose structured - # tool_calls instead of raw XTML in message.content. - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "kimi_k3" - reasoning-parser: "kimi_k3" - enable-auto-tool-choice: true - router-mode: "kv" - router-kv-events: true - router-temperature: "0" - router-min-initial-workers: 1 - kv-cache-block-size: 64 - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 16 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - NCCL_P2P_LEVEL: "NVL" - NVIDIA_GDRCOPY: "1" - PYTORCH_ALLOC_CONF: "expandable_segments:True" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-kimi-k3-gb200-tep16-{job_id}" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 16 - pipeline-parallel-size: 1 - enable-expert-parallel: true - trust-remote-code: true - load-format: "fastsafetensors" - safetensors-load-strategy: "lazy" - kv-cache-dtype: "fp8" - attention-backend: "FLASHINFER_MLA" - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - moe-backend: "flashinfer_trtllm" - kda-prefill-backend: "flashkda" - kernel-config: '{"enable_cutedsl_warmup":true}' - all2all-backend: "flashinfer_nvlink_one_sided" - gpu-memory-utilization: 0.92 - max-num-seqs: 32 - max-num-batched-tokens: 8192 - speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - compilation-config: '{"cudagraph_capture_sizes":[3,6,9,12,15,18,21,24,27,30,33,36,39,42,45,48,51,54,57,60,63,66,69,72,75,78,81,84,87,90,93,96],"pass_config":{"fuse_allreduce_rms":false}}' - block-size: 64 - language-model-only: true - disable-custom-all-reduce: true - enable-prefix-caching: true - scheduler-cls: "vllm.v1.core.sched.async_scheduler.AsyncScheduler" - dyn-tool-call-parser: "kimi_k3" - reasoning-parser: "kimi_k3" - dyn-reasoning-parser: "kimi_k3" - no-enable-flashinfer-autotune: true - -sbatch_directives: - cpus-per-task: "144" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp16-latency.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp16-latency.yaml deleted file mode 100644 index a838dc8f73..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp16-latency.yaml +++ /dev/null @@ -1,149 +0,0 @@ -schema: 2 -name: "kimi-k3-vllm-agg-gb200-tp16-latency-agentic" - -# Day-0 GB200 translation of the official latency-oriented multi_node_tp -# profile. TP16 spans four GB200 nodes and uses the official FP8 KV cache. -# https://recipes.vllm.ai/moonshotai/Kimi-K3?hardware=gb200&nodes=4&strategy=multi_node_tp - -model: - path: "kimi-k3" - container: "vllm/vllm-openai:kimi-k3" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:kimi-k3" - frameworks: - dynamo: "1.3.0" - -dynamo: - install: true - - source: - rev: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" -setup_script: kimik3-dspark-config-compat.sh - -environment: - ETCD_LEASE_TTL: "7200" - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - enable_multiple_frontends: false - args: - # Use vLLM's Kimi K3 parser so OpenAI responses expose structured - # tool_calls instead of raw XTML in message.content. - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "kimi_k3" - reasoning-parser: "kimi_k3" - enable-auto-tool-choice: true - router-mode: "kv" - router-kv-events: true - router-temperature: "0" - router-min-initial-workers: 1 - kv-cache-block-size: 64 - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 16 - - env: - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_RPC_TIMEOUT: "600000" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "rc,cuda_copy" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - NCCL_P2P_LEVEL: "NVL" - NVIDIA_GDRCOPY: "1" - PYTORCH_ALLOC_CONF: "expandable_segments:True" - DG_JIT_CACHE_DIR: "/tmp/dg-cache-kimi-k3-gb200-tp16-{job_id}" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 16 - pipeline-parallel-size: 1 - trust-remote-code: true - load-format: "fastsafetensors" - safetensors-load-strategy: "lazy" - kv-cache-dtype: "fp8" - attention-backend: "FLASHINFER_MLA" - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - moe-backend: "flashinfer_trtllm" - kda-prefill-backend: "flashkda" - kernel-config: '{"enable_cutedsl_warmup":true}' - gpu-memory-utilization: 0.92 - max-num-seqs: 8 - max-num-batched-tokens: 8192 - speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - compilation-config: '{"cudagraph_capture_sizes":[3,6,9,12,15,18,21,24],"pass_config":{"fuse_allreduce_rms":false}}' - block-size: 64 - language-model-only: true - disable-custom-all-reduce: true - enable-prefix-caching: true - scheduler-cls: "vllm.v1.core.sched.async_scheduler.AsyncScheduler" - dyn-tool-call-parser: "kimi_k3" - reasoning-parser: "kimi_k3" - dyn-reasoning-parser: "kimi_k3" - no-enable-flashinfer-autotune: true - -sbatch_directives: - cpus-per-task: "144" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c16.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c16.yaml deleted file mode 100644 index 33f3115102..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c16.yaml +++ /dev/null @@ -1,146 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-gb200-tp8pp2-mooncake-c16-agentic" - -model: - path: "kimi-k3" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - interval_seconds: 10 - max_attempts: 720 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 16 - - env: - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_RPC_TIMEOUT: "600000" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - NCCL_CUMEM_ENABLE: "1" - TILELANG_CLEANUP_TEMP_FILES: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - pipeline-parallel-size: 2 - decode-context-parallel-size: 8 - dcp-comm-backend: a2a - trust-remote-code: true - load-format: fastsafetensors - moe-backend: auto - gpu-memory-utilization: 0.87 - no-enable-flashinfer-autotune: true - max-model-len: 1048576 - kv-cache-dtype: fp8 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - attention-backend: TOKENSPEED_MLA - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - max-num-seqs: 32 - max-num-batched-tokens: 8192 - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' - prefix-match-unit: 128 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,24,40,56,64,80,104,128,256,512,1024,2048,4096,8192]}' - -services: - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -sbatch_directives: - segment: "1" - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - client_placement: head - concurrencies: [16] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - ENABLE_AGENTX_POWER: "1" - REQUIRE_POWER: "1" - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_SERVER_METRICS_URLS: "http://localhost:8000/metrics" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c32.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c32.yaml deleted file mode 100644 index cf98995eec..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c32.yaml +++ /dev/null @@ -1,146 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-gb200-tp8pp2-mooncake-c32-agentic" - -model: - path: "kimi-k3" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - interval_seconds: 10 - max_attempts: 720 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 16 - - env: - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_RPC_TIMEOUT: "600000" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - NCCL_CUMEM_ENABLE: "1" - TILELANG_CLEANUP_TEMP_FILES: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - pipeline-parallel-size: 2 - decode-context-parallel-size: 8 - dcp-comm-backend: a2a - trust-remote-code: true - load-format: fastsafetensors - moe-backend: auto - gpu-memory-utilization: 0.87 - no-enable-flashinfer-autotune: true - max-model-len: 1048576 - kv-cache-dtype: fp8 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - attention-backend: TOKENSPEED_MLA - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - max-num-seqs: 64 - max-num-batched-tokens: 8192 - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' - prefix-match-unit: 128 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,7,18,34,53,64,75,100,128,256,512,1024,2048,4096,8192]}' - -services: - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -sbatch_directives: - segment: "1" - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - client_placement: head - concurrencies: [32] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - ENABLE_AGENTX_POWER: "1" - REQUIRE_POWER: "1" - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_SERVER_METRICS_URLS: "http://localhost:8000/metrics" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c48.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c48.yaml deleted file mode 100644 index 57498c8969..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c48.yaml +++ /dev/null @@ -1,145 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-gb200-tp8pp2-mooncake-c48-agentic" - -model: - path: "kimi-k3" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - interval_seconds: 10 - max_attempts: 720 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 16 - - env: - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_RPC_TIMEOUT: "600000" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - NCCL_CUMEM_ENABLE: "1" - TILELANG_CLEANUP_TEMP_FILES: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - pipeline-parallel-size: 2 - decode-context-parallel-size: 8 - dcp-comm-backend: a2a - trust-remote-code: true - load-format: fastsafetensors - moe-backend: auto - gpu-memory-utilization: 0.87 - no-enable-flashinfer-autotune: true - max-model-len: 1048576 - kv-cache-dtype: fp8 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - attention-backend: TOKENSPEED_MLA - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - max-num-seqs: 96 - max-num-batched-tokens: 8192 - prefix-match-unit: 128 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,7,18,34,53,64,75,100,128,256,512,1024,2048,4096,8192]}' - -services: - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -sbatch_directives: - segment: "1" - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - client_placement: head - concurrencies: [48] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - ENABLE_AGENTX_POWER: "1" - REQUIRE_POWER: "1" - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_SERVER_METRICS_URLS: "http://localhost:8000/metrics" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c72.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c72.yaml deleted file mode 100644 index c7e1da7937..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c72.yaml +++ /dev/null @@ -1,145 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-gb200-tp8pp2-mooncake-c72-agentic" - -model: - path: "kimi-k3" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - interval_seconds: 10 - max_attempts: 720 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 16 - - env: - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_RPC_TIMEOUT: "600000" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - NCCL_CUMEM_ENABLE: "1" - TILELANG_CLEANUP_TEMP_FILES: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - pipeline-parallel-size: 2 - decode-context-parallel-size: 8 - dcp-comm-backend: a2a - trust-remote-code: true - load-format: fastsafetensors - moe-backend: auto - gpu-memory-utilization: 0.87 - no-enable-flashinfer-autotune: true - max-model-len: 1048576 - kv-cache-dtype: fp8 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - attention-backend: TOKENSPEED_MLA - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - max-num-seqs: 144 - max-num-batched-tokens: 8192 - prefix-match-unit: 128 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,7,18,34,53,64,75,100,128,256,512,1024,2048,4096,8192]}' - -services: - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -sbatch_directives: - segment: "1" - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - client_placement: head - concurrencies: [72] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - ENABLE_AGENTX_POWER: "1" - REQUIRE_POWER: "1" - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_SERVER_METRICS_URLS: "http://localhost:8000/metrics" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c96.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c96.yaml deleted file mode 100644 index 2bb53eaee4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c96.yaml +++ /dev/null @@ -1,145 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-gb200-tp8pp2-mooncake-c96-agentic" - -model: - path: "kimi-k3" - container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - interval_seconds: 10 - max_attempts: 720 - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 16 - - env: - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - MC_GID_INDEX: "3" - MC_STORE_MEMCPY: "1" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_SLICE_SIZE: "1048576" - MC_WORKERS_PER_CTX: "4" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - VLLM_RPC_TIMEOUT: "600000" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: "0" - NCCL_CUMEM_ENABLE: "1" - TILELANG_CLEANUP_TEMP_FILES: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - pipeline-parallel-size: 2 - decode-context-parallel-size: 8 - dcp-comm-backend: a2a - trust-remote-code: true - load-format: fastsafetensors - moe-backend: auto - gpu-memory-utilization: 0.87 - no-enable-flashinfer-autotune: true - max-model-len: 1048576 - kv-cache-dtype: fp8 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' - attention-backend: TOKENSPEED_MLA - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - max-num-seqs: 192 - max-num-batched-tokens: 8192 - prefix-match-unit: 128 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,7,18,34,53,64,75,100,128,256,512,1024,2048,4096,8192]}' - -services: - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -sbatch_directives: - segment: "1" - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - client_placement: head - concurrencies: [96] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - ENABLE_AGENTX_POWER: "1" - REQUIRE_POWER: "1" - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_SERVER_METRICS_URLS: "http://localhost:8000/metrics" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..66d1c4ac0d --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml @@ -0,0 +1,1186 @@ +# srt-slurm recipes for kimik3/vllm/gb200-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: kimi-k3 + precision: fp4 + identity: + model: + repo: moonshotai/Kimi-K3 + container: {} + dynamo: {} + slurm: {} + health_check: + interval_seconds: 10 + resources: + gpu_type: gb200 + gpus_per_node: 4 + frontend: + enable_multiple_frontends: false + engine: + type: vllm + connector: null + roles: + agg: + nodes: 4 + workers: 1 + gpus: 16 + env: + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_RPC_TIMEOUT: '600000' + args: + served-model-name: moonshotai/Kimi-K3 + trust-remote-code: true + language-model-only: true + reasoning-parser: kimi_k3 + no-enable-flashinfer-autotune: true + enable-prefix-caching: true + kv-cache-dtype: fp8 + attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' + sbatch_directives: {} + srun_options: + container-remap-root: '' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: '300' + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + +# GB200 TP16/DCP16 aggregate profile with DSpark K=4 and max-num-seqs 2. +override_agg_dcp16_dspark4_maxseq2_mooncake: + name: kimi-k3-vllm-agg-gb200-dcp16-dspark4-maxseq2-mooncake-agentic + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-75c2eef + identity: + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-75c2eef + frameworks: + dynamo: ba83080ecd31c1ce918559e576d3c5bc9e092ff1 + dynamo: + install: true + source: + rev: ba83080ecd31c1ce918559e576d3c5bc9e092ff1 + slurm: + time_limit: 04:00:00 + health_check: + max_attempts: 720 + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + args: [--eviction_high_watermark_ratio=0.95, --eviction_ratio=0.10] + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 96GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + type: dynamo + args: + dyn-chat-processor: vllm + trust-remote-code: true + tool-call-parser: kimi_k3 + reasoning-parser: kimi_k3 + enable-auto-tool-choice: true + router-mode: random + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + DYN_TOKENIZER_CACHE_BYTES: '8589934592' + roles: + agg: + env: + # Preserve the measured collective fallback when moving to the built image. + VLLM_USE_DIRECT_DCP_A2A: '0' + VLLM_USE_DIRECT_DCP_Q_GATHER: '0' + VLLM_USE_DIRECT_DCP_KV_GATHER: '0' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0' + PYTHONNOUSERSITE: '1' + TORCH_CUDA_ARCH_LIST: '10.0' + PYTHONHASHSEED: '42' + VLLM_HTTP_TIMEOUT_KEEP_ALIVE: '900' + MC_GID_INDEX: '3' + MC_STORE_MEMCPY: '1' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_SLICE_SIZE: '1048576' + MC_WORKERS_PER_CTX: '4' + WITH_NVIDIA_PEERMEM: '0' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + args: + tensor-parallel-size: 16 + decode-context-parallel-size: 16 + dcp-comm-backend: a2a + max-num-seqs: 2 + max-num-batched-tokens: 8192 + dyn-tool-call-parser: kimi_k3 + dyn-reasoning-parser: kimi_k3 + load-format: safetensors + safetensors-load-strategy: lazy + moe-backend: auto + enable-cumem-allocator: true + prefix-match-unit: 128 + stream-interval: 10 + attention-backend: TOKENSPEED_MLA + # Throughput jobs inject the committed K=4 golden AL (3.36); EVAL_ONLY + # preserves this real target-verification configuration. + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + max-cudagraph-capture-size: 1024 + kv-cache-memory: 10737418240 + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' + sbatch_directives: + mem: '0' + cpus-per-task: '144' + comment: '''{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will take extra loading time."}}''' + srun_options: + mem: '0' + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + collector_join_timeout_seconds: 12 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + benchmark: + client_placement: head + concurrencies: [1, 2, 4, 8, 16] + env: + ENABLE_AGENTX_POWER: '1' + REQUIRE_POWER: '1' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' + AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' + +# GB200 TP16/DCP16 aggregate profile without speculative decoding. +override_agg_dcp16_nospec_mooncake: + name: kimi-k3-vllm-agg-gb200-dcp16-nospec-mooncake-agentic + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-75c2eef + identity: + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-75c2eef + frameworks: + dynamo: ba83080ecd31c1ce918559e576d3c5bc9e092ff1 + dynamo: + install: true + source: + rev: ba83080ecd31c1ce918559e576d3c5bc9e092ff1 + slurm: + time_limit: 04:00:00 + health_check: + max_attempts: 720 + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + args: [--eviction_high_watermark_ratio=0.95, --eviction_ratio=0.10] + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 96GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + type: dynamo + args: + dyn-chat-processor: vllm + trust-remote-code: true + tool-call-parser: kimi_k3 + reasoning-parser: kimi_k3 + enable-auto-tool-choice: true + router-mode: random + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + DYN_TOKENIZER_CACHE_BYTES: '8589934592' + roles: + agg: + env: + # Preserve the measured collective fallback when moving to the built image. + VLLM_USE_DIRECT_DCP_A2A: '0' + VLLM_USE_DIRECT_DCP_Q_GATHER: '0' + VLLM_USE_DIRECT_DCP_KV_GATHER: '0' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0' + PYTHONNOUSERSITE: '1' + TORCH_CUDA_ARCH_LIST: '10.0' + PYTHONHASHSEED: '42' + VLLM_HTTP_TIMEOUT_KEEP_ALIVE: '900' + MC_GID_INDEX: '3' + MC_STORE_MEMCPY: '1' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_SLICE_SIZE: '1048576' + MC_WORKERS_PER_CTX: '4' + WITH_NVIDIA_PEERMEM: '0' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + args: + tensor-parallel-size: 16 + decode-context-parallel-size: 16 + dcp-comm-backend: a2a + max-num-seqs: 1000 + max-num-batched-tokens: 16384 + dyn-tool-call-parser: kimi_k3 + dyn-reasoning-parser: kimi_k3 + load-format: safetensors + safetensors-load-strategy: lazy + moe-backend: auto + prefix-match-unit: 128 + stream-interval: 10 + attention-backend: FLASHINFER_MLA + max-cudagraph-capture-size: 1024 + kv-cache-memory: 10737418240 + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' + sbatch_directives: + mem: '0' + cpus-per-task: '144' + comment: '''{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will take extra loading time."}}''' + srun_options: + mem: '0' + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + collector_join_timeout_seconds: 12 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + benchmark: + client_placement: head + concurrencies: [8, 40, 48] + env: + ENABLE_AGENTX_POWER: '1' + REQUIRE_POWER: '1' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' + AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' + +override_agg_dep16_vllm_simple_offload: + name: kimi-k3-vllm-agg-gb200-dep16-throughput-vllm-simple-offload-agentic + # High-concurrency host-DRAM KV-offload variant of the official throughput- + # oriented multi_node_dep profile. TP4 x DP4 gives EP16 across four four-GPU + # GB200 nodes. Each TP rank receives a 128 GiB CPU KV pool (512 GiB per node). + # https://recipes.vllm.ai/moonshotai/Kimi-K3?hardware=gb200&nodes=4&strategy=multi_node_dep + model: + container: vllm/vllm-openai:kimi-k3 + identity: + container: + image: vllm/vllm-openai:kimi-k3 + frameworks: + dynamo: 1.3.0 + dynamo: + install: true + source: + rev: ba83080ecd31c1ce918559e576d3c5bc9e092ff1 + slurm: + time_limit: '12:00:00' + health_check: + max_attempts: 2160 + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + type: dynamo + args: + # Use vLLM's Kimi K3 parser so OpenAI responses expose structured + # tool_calls instead of raw XTML in message.content. + dyn-chat-processor: vllm + trust-remote-code: true + tool-call-parser: kimi_k3 + reasoning-parser: kimi_k3 + enable-auto-tool-choice: true + router-mode: kv + router-kv-events: true + router-temperature: '0' + router-min-initial-workers: 1 + kv-cache-block-size: 64 + roles: + agg: + env: + VLLM_ENGINE_READY_TIMEOUT_S: '7200' + PYTHONHASHSEED: '42' + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_FLASHINFER_ALLREDUCE_BACKEND: mnnvl + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + NCCL_P2P_LEVEL: NVL + NVIDIA_GDRCOPY: '1' + PYTORCH_ALLOC_CONF: expandable_segments:True + DG_JIT_CACHE_DIR: /tmp/dg-cache-kimi-k3-gb200-dep16-offload-{job_id} + args: + tensor-parallel-size: 4 + # The largest offload point is c384 / DP4 = 96 sequences per engine. + # Capture even sequence counts: all configured DP4 steady-state batch + # sizes are exact hits, while odd loads pad by at most one sequence. + max-num-seqs: 96 + max-num-batched-tokens: 16384 + dyn-tool-call-parser: kimi_k3 + dyn-reasoning-parser: kimi_k3 + # FlashInfer's larger TP4 MoE representation leaves too little transient + # HBM for fastsafetensors' GPU-staging path when DSpark is loaded. + load-format: safetensors + safetensors-load-strategy: lazy + # DeepGEMM mega-MoE grid barriers time out under DSpark MRV2 with the + # one-sided all-to-all worker; use the supported Kimi K3 FlashInfer path. + moe-backend: flashinfer_trtllm + attention-backend: FLASHINFER_MLA + speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":549755813888,"cpu_bytes_to_use_per_rank":137438953472,"lazy_offload":false}}' + pipeline-parallel-size: 1 + data-parallel-size: 4 + data-parallel-rpc-port: 13345 + enable-expert-parallel: true + kda-prefill-backend: flashkda + kernel-config: '{"enable_cutedsl_warmup":true}' + all2all-backend: flashinfer_nvlink_one_sided + gpu-memory-utilization: 0.94 + compilation-config: '{"cudagraph_mode":"PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,42,48,54,60,66,72,78,84,90,96,102,108,114,120,126,132,138,144,150,156,162,168,174,180,186,192,198,204,210,216,222,228,234,240,246,252,258,264,270,276,282,288]}' + block-size: 64 + disable-custom-all-reduce: true + scheduler-cls: vllm.v1.core.sched.async_scheduler.AsyncScheduler + sbatch_directives: + mem: '0' + cpus-per-task: '144' + benchmark: + env: + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + AGENTIC_WARMUP_GRACE_PERIOD: '3600' + setup_script: kimik3-dspark-config-compat.sh + environment: + ETCD_LEASE_TTL: '7200' + +override_agg_dep16: + name: kimi-k3-vllm-agg-gb200-dep16-throughput-agentic + # https://recipes.vllm.ai/moonshotai/Kimi-K3?hardware=gb200&nodes=4&strategy=multi_node_dep + # Day-0 GB200 translation of the official throughput-oriented multi_node_dep + # profile. TP4 x DP4 gives EP16 across four four-GPU GB200 nodes. + model: + container: vllm/vllm-openai:kimi-k3 + identity: + container: + image: vllm/vllm-openai:kimi-k3 + frameworks: + dynamo: 1.3.0 + dynamo: + install: true + source: + rev: ba83080ecd31c1ce918559e576d3c5bc9e092ff1 + slurm: + time_limit: '12:00:00' + health_check: + max_attempts: 2160 + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + type: dynamo + args: + # Use vLLM's Kimi K3 parser so OpenAI responses expose structured + # tool_calls instead of raw XTML in message.content. + dyn-chat-processor: vllm + trust-remote-code: true + tool-call-parser: kimi_k3 + reasoning-parser: kimi_k3 + enable-auto-tool-choice: true + router-mode: kv + router-kv-events: true + router-temperature: '0' + router-min-initial-workers: 1 + kv-cache-block-size: 64 + roles: + agg: + env: + VLLM_ENGINE_READY_TIMEOUT_S: '7200' + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_FLASHINFER_ALLREDUCE_BACKEND: mnnvl + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + NCCL_P2P_LEVEL: NVL + NVIDIA_GDRCOPY: '1' + PYTORCH_ALLOC_CONF: expandable_segments:True + DG_JIT_CACHE_DIR: /tmp/dg-cache-kimi-k3-gb200-dep16-{job_id} + args: + tensor-parallel-size: 4 + # The largest regular DEP point is c256 / DP4 = 64 sequences per engine. + # Capturing 128 sequence slots consumes 10.9 GiB and leaves too little + # runtime workspace for FlashInfer's MXFP4 MoE kernel. + max-num-seqs: 64 + max-num-batched-tokens: 16384 + dyn-tool-call-parser: kimi_k3 + dyn-reasoning-parser: kimi_k3 + # FlashInfer's larger TP4 MoE representation leaves too little transient + # HBM for fastsafetensors' GPU-staging path when DSpark is loaded. + load-format: safetensors + safetensors-load-strategy: lazy + # DeepGEMM mega-MoE grid barriers time out under DSpark MRV2 with the + # one-sided all-to-all worker; use the supported Kimi K3 FlashInfer path. + moe-backend: flashinfer_trtllm + attention-backend: FLASHINFER_MLA + speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + pipeline-parallel-size: 1 + data-parallel-size: 4 + data-parallel-rpc-port: 13345 + enable-expert-parallel: true + kda-prefill-backend: flashkda + kernel-config: '{"enable_cutedsl_warmup":true}' + all2all-backend: flashinfer_nvlink_one_sided + gpu-memory-utilization: 0.94 + compilation-config: '{"cudagraph_mode":"PIECEWISE","cudagraph_capture_sizes":[3,6,9,12,15,18,21,24,27,30,33,36,39,42,45,48,51,54,57,60,63,66,69,72,75,78,81,84,87,90,93,96,99,102,105,108,111,114,117,120,123,126,129,132,135,138,141,144,147,150,153,156,159,162,165,168,171,174,177,180,183,186,189,192]}' + block-size: 64 + disable-custom-all-reduce: true + scheduler-cls: vllm.v1.core.sched.async_scheduler.AsyncScheduler + sbatch_directives: + mem: '0' + cpus-per-task: '144' + benchmark: + env: + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + AGENTIC_WARMUP_GRACE_PERIOD: '3600' + setup_script: kimik3-dspark-config-compat.sh + environment: + ETCD_LEASE_TTL: '7200' + +override_agg_tep16_balanced: + name: kimi-k3-vllm-agg-gb200-tep16-balanced-agentic + # Day-0 GB200 translation of the official balanced multi_node_tep profile. + # Dense layers and MoE experts are sharded across 16 GPUs on four GB200 nodes + # with the official FP8 KV cache. + # https://recipes.vllm.ai/moonshotai/Kimi-K3?hardware=gb200&nodes=4&strategy=multi_node_tep + model: + container: vllm/vllm-openai:kimi-k3 + identity: + container: + image: vllm/vllm-openai:kimi-k3 + frameworks: + dynamo: 1.3.0 + dynamo: + install: true + source: + rev: ba83080ecd31c1ce918559e576d3c5bc9e092ff1 + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 2160 + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + type: dynamo + args: + # Use vLLM's Kimi K3 parser so OpenAI responses expose structured + # tool_calls instead of raw XTML in message.content. + dyn-chat-processor: vllm + trust-remote-code: true + tool-call-parser: kimi_k3 + reasoning-parser: kimi_k3 + enable-auto-tool-choice: true + router-mode: kv + router-kv-events: true + router-temperature: '0' + router-min-initial-workers: 1 + kv-cache-block-size: 64 + roles: + agg: + env: + VLLM_ENGINE_READY_TIMEOUT_S: '7200' + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_FLASHINFER_ALLREDUCE_BACKEND: mnnvl + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + NCCL_P2P_LEVEL: NVL + NVIDIA_GDRCOPY: '1' + PYTORCH_ALLOC_CONF: expandable_segments:True + DG_JIT_CACHE_DIR: /tmp/dg-cache-kimi-k3-gb200-tep16-{job_id} + args: + tensor-parallel-size: 16 + max-num-seqs: 32 + max-num-batched-tokens: 8192 + dyn-tool-call-parser: kimi_k3 + dyn-reasoning-parser: kimi_k3 + load-format: fastsafetensors + safetensors-load-strategy: lazy + moe-backend: flashinfer_trtllm + attention-backend: FLASHINFER_MLA + speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + pipeline-parallel-size: 1 + enable-expert-parallel: true + kda-prefill-backend: flashkda + kernel-config: '{"enable_cutedsl_warmup":true}' + all2all-backend: flashinfer_nvlink_one_sided + gpu-memory-utilization: 0.92 + compilation-config: '{"cudagraph_capture_sizes":[3,6,9,12,15,18,21,24,27,30,33,36,39,42,45,48,51,54,57,60,63,66,69,72,75,78,81,84,87,90,93,96],"pass_config":{"fuse_allreduce_rms":false}}' + block-size: 64 + disable-custom-all-reduce: true + scheduler-cls: vllm.v1.core.sched.async_scheduler.AsyncScheduler + sbatch_directives: + mem: '0' + cpus-per-task: '144' + benchmark: + env: + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + setup_script: kimik3-dspark-config-compat.sh + environment: + ETCD_LEASE_TTL: '7200' + +override_agg_tp16_latency: + name: kimi-k3-vllm-agg-gb200-tp16-latency-agentic + # Day-0 GB200 translation of the official latency-oriented multi_node_tp + # profile. TP16 spans four GB200 nodes and uses the official FP8 KV cache. + # https://recipes.vllm.ai/moonshotai/Kimi-K3?hardware=gb200&nodes=4&strategy=multi_node_tp + model: + container: vllm/vllm-openai:kimi-k3 + identity: + container: + image: vllm/vllm-openai:kimi-k3 + frameworks: + dynamo: 1.3.0 + dynamo: + install: true + source: + rev: ba83080ecd31c1ce918559e576d3c5bc9e092ff1 + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 2160 + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + type: dynamo + args: + # Use vLLM's Kimi K3 parser so OpenAI responses expose structured + # tool_calls instead of raw XTML in message.content. + dyn-chat-processor: vllm + trust-remote-code: true + tool-call-parser: kimi_k3 + reasoning-parser: kimi_k3 + enable-auto-tool-choice: true + router-mode: kv + router-kv-events: true + router-temperature: '0' + router-min-initial-workers: 1 + kv-cache-block-size: 64 + roles: + agg: + env: + VLLM_ENGINE_READY_TIMEOUT_S: '7200' + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_FLASHINFER_ALLREDUCE_BACKEND: mnnvl + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: rc,cuda_copy + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + NCCL_P2P_LEVEL: NVL + NVIDIA_GDRCOPY: '1' + PYTORCH_ALLOC_CONF: expandable_segments:True + DG_JIT_CACHE_DIR: /tmp/dg-cache-kimi-k3-gb200-tp16-{job_id} + args: + tensor-parallel-size: 16 + max-num-seqs: 8 + max-num-batched-tokens: 8192 + dyn-tool-call-parser: kimi_k3 + dyn-reasoning-parser: kimi_k3 + load-format: fastsafetensors + safetensors-load-strategy: lazy + moe-backend: flashinfer_trtllm + attention-backend: FLASHINFER_MLA + speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + pipeline-parallel-size: 1 + kda-prefill-backend: flashkda + kernel-config: '{"enable_cutedsl_warmup":true}' + gpu-memory-utilization: 0.92 + compilation-config: '{"cudagraph_capture_sizes":[3,6,9,12,15,18,21,24],"pass_config":{"fuse_allreduce_rms":false}}' + block-size: 64 + disable-custom-all-reduce: true + scheduler-cls: vllm.v1.core.sched.async_scheduler.AsyncScheduler + sbatch_directives: + mem: '0' + cpus-per-task: '144' + benchmark: + env: + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + setup_script: kimik3-dspark-config-compat.sh + environment: + ETCD_LEASE_TTL: '7200' + +override_agg_tp8pp2_mooncake_c16: + name: kimik3-vllm-agg-gb200-tp8pp2-mooncake-c16-agentic + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad + identity: + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad + dynamo: + install: false + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 720 + services: + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 200GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + type: vllm + roles: + agg: + env: + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0' + MC_GID_INDEX: '3' + MC_STORE_MEMCPY: '1' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_SLICE_SIZE: '1048576' + MC_WORKERS_PER_CTX: '4' + WITH_NVIDIA_PEERMEM: '0' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + VLLM_LOG_STATS_INTERVAL: '1' + NCCL_CUMEM_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_SERVER_DEV_MODE: '1' + TILELANG_CLEANUP_TEMP_FILES: '1' + UCX_MEMTYPE_REG_WHOLE: n + args: + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + dcp-comm-backend: a2a + max-num-seqs: 32 + max-num-batched-tokens: 8192 + load-format: fastsafetensors + moe-backend: auto + prefix-match-unit: 128 + attention-backend: TOKENSPEED_MLA + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' + pipeline-parallel-size: 2 + gpu-memory-utilization: 0.87 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,24,40,56,64,80,104,128,256,512,1024,2048,4096,8192]}' + max-model-len: 1048576 + enable-prompt-tokens-details: true + enable-auto-tool-choice: true + tool-call-parser: kimi_k3 + sbatch_directives: + segment: '1' + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + collector_join_timeout_seconds: 12 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + benchmark: + client_placement: head + concurrencies: [16] + env: + ENABLE_AGENTX_POWER: '1' + REQUIRE_POWER: '1' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' + AIPERF_SERVER_METRICS_URLS: http://localhost:8000/metrics + +override_agg_tp8pp2_mooncake_c32: + name: kimik3-vllm-agg-gb200-tp8pp2-mooncake-c32-agentic + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad + identity: + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad + dynamo: + install: false + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 720 + services: + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 200GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + type: vllm + roles: + agg: + env: + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0' + MC_GID_INDEX: '3' + MC_STORE_MEMCPY: '1' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_SLICE_SIZE: '1048576' + MC_WORKERS_PER_CTX: '4' + WITH_NVIDIA_PEERMEM: '0' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + VLLM_LOG_STATS_INTERVAL: '1' + NCCL_CUMEM_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_SERVER_DEV_MODE: '1' + TILELANG_CLEANUP_TEMP_FILES: '1' + UCX_MEMTYPE_REG_WHOLE: n + args: + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + dcp-comm-backend: a2a + max-num-seqs: 64 + max-num-batched-tokens: 8192 + load-format: fastsafetensors + moe-backend: auto + prefix-match-unit: 128 + attention-backend: TOKENSPEED_MLA + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"method":"dspark","attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic"}' + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' + pipeline-parallel-size: 2 + gpu-memory-utilization: 0.87 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,7,18,34,53,64,75,100,128,256,512,1024,2048,4096,8192]}' + max-model-len: 1048576 + enable-prompt-tokens-details: true + enable-auto-tool-choice: true + tool-call-parser: kimi_k3 + sbatch_directives: + segment: '1' + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + collector_join_timeout_seconds: 12 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + benchmark: + client_placement: head + concurrencies: [32] + env: + ENABLE_AGENTX_POWER: '1' + REQUIRE_POWER: '1' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' + AIPERF_SERVER_METRICS_URLS: http://localhost:8000/metrics + +override_agg_tp8pp2_mooncake_c48: + name: kimik3-vllm-agg-gb200-tp8pp2-mooncake-c48-agentic + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad + identity: + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad + dynamo: + install: false + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 720 + services: + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 200GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + type: vllm + roles: + agg: + env: + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0' + MC_GID_INDEX: '3' + MC_STORE_MEMCPY: '1' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_SLICE_SIZE: '1048576' + MC_WORKERS_PER_CTX: '4' + WITH_NVIDIA_PEERMEM: '0' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + VLLM_LOG_STATS_INTERVAL: '1' + NCCL_CUMEM_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_SERVER_DEV_MODE: '1' + TILELANG_CLEANUP_TEMP_FILES: '1' + UCX_MEMTYPE_REG_WHOLE: n + args: + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + dcp-comm-backend: a2a + max-num-seqs: 96 + max-num-batched-tokens: 8192 + load-format: fastsafetensors + moe-backend: auto + prefix-match-unit: 128 + attention-backend: TOKENSPEED_MLA + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' + pipeline-parallel-size: 2 + gpu-memory-utilization: 0.87 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,7,18,34,53,64,75,100,128,256,512,1024,2048,4096,8192]}' + max-model-len: 1048576 + enable-prompt-tokens-details: true + enable-auto-tool-choice: true + tool-call-parser: kimi_k3 + sbatch_directives: + segment: '1' + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + collector_join_timeout_seconds: 12 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + benchmark: + client_placement: head + concurrencies: [48] + env: + ENABLE_AGENTX_POWER: '1' + REQUIRE_POWER: '1' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' + AIPERF_SERVER_METRICS_URLS: http://localhost:8000/metrics + +override_agg_tp8pp2_mooncake_c72: + name: kimik3-vllm-agg-gb200-tp8pp2-mooncake-c72-agentic + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad + identity: + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad + dynamo: + install: false + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 720 + services: + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 200GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + type: vllm + roles: + agg: + env: + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0' + MC_GID_INDEX: '3' + MC_STORE_MEMCPY: '1' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_SLICE_SIZE: '1048576' + MC_WORKERS_PER_CTX: '4' + WITH_NVIDIA_PEERMEM: '0' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + VLLM_LOG_STATS_INTERVAL: '1' + NCCL_CUMEM_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_SERVER_DEV_MODE: '1' + TILELANG_CLEANUP_TEMP_FILES: '1' + UCX_MEMTYPE_REG_WHOLE: n + args: + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + dcp-comm-backend: a2a + max-num-seqs: 144 + max-num-batched-tokens: 8192 + load-format: fastsafetensors + moe-backend: auto + prefix-match-unit: 128 + attention-backend: TOKENSPEED_MLA + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' + pipeline-parallel-size: 2 + gpu-memory-utilization: 0.87 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,7,18,34,53,64,75,100,128,256,512,1024,2048,4096,8192]}' + max-model-len: 1048576 + enable-prompt-tokens-details: true + enable-auto-tool-choice: true + tool-call-parser: kimi_k3 + sbatch_directives: + segment: '1' + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + collector_join_timeout_seconds: 12 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + benchmark: + client_placement: head + concurrencies: [72] + env: + ENABLE_AGENTX_POWER: '1' + REQUIRE_POWER: '1' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' + AIPERF_SERVER_METRICS_URLS: http://localhost:8000/metrics + +override_agg_tp8pp2_mooncake_c96: + name: kimik3-vllm-agg-gb200-tp8pp2-mooncake-c96-agentic + model: + container: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad + identity: + container: + image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-728d3ad + dynamo: + install: false + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 720 + services: + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 200GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + type: vllm + roles: + agg: + env: + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0' + MC_GID_INDEX: '3' + MC_STORE_MEMCPY: '1' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_SLICE_SIZE: '1048576' + MC_WORKERS_PER_CTX: '4' + WITH_NVIDIA_PEERMEM: '0' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + VLLM_LOG_STATS_INTERVAL: '1' + NCCL_CUMEM_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_SERVER_DEV_MODE: '1' + TILELANG_CLEANUP_TEMP_FILES: '1' + UCX_MEMTYPE_REG_WHOLE: n + args: + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + dcp-comm-backend: a2a + max-num-seqs: 192 + max-num-batched-tokens: 8192 + load-format: fastsafetensors + moe-backend: auto + prefix-match-unit: 128 + attention-backend: TOKENSPEED_MLA + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' + pipeline-parallel-size: 2 + gpu-memory-utilization: 0.87 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,7,18,34,53,64,75,100,128,256,512,1024,2048,4096,8192]}' + max-model-len: 1048576 + enable-prompt-tokens-details: true + enable-auto-tool-choice: true + tool-call-parser: kimi_k3 + sbatch_directives: + segment: '1' + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + collector_join_timeout_seconds: 12 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + benchmark: + client_placement: head + concurrencies: [96] + env: + ENABLE_AGENTX_POWER: '1' + REQUIRE_POWER: '1' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' + AIPERF_SERVER_METRICS_URLS: http://localhost:8000/metrics diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/agg-dcp8-dspark4-mooncake.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/agg-dcp8-dspark4-mooncake.yaml deleted file mode 100644 index bac87abe62..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/agg-dcp8-dspark4-mooncake.yaml +++ /dev/null @@ -1,165 +0,0 @@ -schema: 2 -name: "kimi-k3-vllm-agg-gb300-dcp8-dspark4-mooncake-agentic" - -model: - path: "moonshotai/Kimi-K3" - container: "vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56" - frameworks: - dynamo: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" - -dynamo: - install: true - - source: - rev: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" -slurm: - time_limit: "04:00:00" - -health_check: - max_attempts: 720 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - args: - - "--default_kv_lease_ttl=60000" - - "--eviction_high_watermark_ratio=0.95" - - "--eviction_ratio=0.10" - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "160GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - - env: - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - # Use the direct DCP a2a/gather kernels rather than the collective - # fallback, matching the B300 arm (kimik3_fp4_b300_vllm_mtp.sh). - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - ETCD_LEASE_TTL: "120" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_USE_RUST_FRONTEND: "1" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - PYTHONNOUSERSITE: "1" - TORCH_CUDA_ARCH_LIST: "10.0" - PYTHONHASHSEED: "42" - VLLM_HTTP_TIMEOUT_KEEP_ALIVE: "900" - MC_SLICE_SIZE: "1048576" - MC_MAX_MR_SIZE: "4294967296" - WITH_NVIDIA_PEERMEM: "0" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - decode-context-parallel-size: 8 - dcp-comm-backend: "a2a" - gpu-memory-utilization: 0.92 - max-num-batched-tokens: 16384 - trust-remote-code: true - language-model-only: true - load-format: "fastsafetensors" - moe-backend: "auto" - enable-flashinfer-autotune: true - enable-prefix-caching: true - prefix-match-unit: 128 - kv-cache-dtype: "fp8" - mamba-ssm-cache-dtype: "bfloat16" - stream-interval: 10 - attention-backend: "TOKENSPEED_MLA" - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - max-cudagraph-capture-size: 1024 - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_offload":false}}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' - -sbatch_directives: - mem: "0" - cpus-per-task: "72" - comment: >- - '{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will - take extra loading time."}}' - -srun_options: - mem: "0" - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 - -benchmark: - type: custom - concurrencies: [48, 52, 56] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - ENABLE_AGENTX_POWER: "1" - REQUIRE_POWER: "1" - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" - placement: - node: head diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/agg-dcp8-dspark7-maxseq2-mooncake.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/agg-dcp8-dspark7-maxseq2-mooncake.yaml deleted file mode 100644 index 113a947f37..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/agg-dcp8-dspark7-maxseq2-mooncake.yaml +++ /dev/null @@ -1,159 +0,0 @@ -schema: 2 -name: "kimi-k3-vllm-agg-gb300-dcp8-dspark7-maxseq2-mooncake-agentic" - -model: - path: "moonshotai/Kimi-K3" - container: "vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56" - frameworks: - dynamo: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" - -dynamo: - install: true - - source: - rev: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" -slurm: - time_limit: "04:00:00" - -health_check: - max_attempts: 720 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - args: - - "--default_kv_lease_ttl=60000" - - "--eviction_high_watermark_ratio=0.95" - - "--eviction_ratio=0.10" - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "160GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - - env: - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - ETCD_LEASE_TTL: "120" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_USE_RUST_FRONTEND: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - MC_SLICE_SIZE: "1048576" - MC_MAX_MR_SIZE: "4294967296" - WITH_NVIDIA_PEERMEM: "0" - VLLM_LOG_STATS_INTERVAL: "1" - - args: - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_offload":false}}' - served-model-name: "moonshotai/Kimi-K3" - load-format: "fastsafetensors" - kv-cache-dtype: "fp8" - mamba-ssm-cache-dtype: "bfloat16" - gpu-memory-utilization: 0.92 - tensor-parallel-size: 8 - decode-context-parallel-size: 8 - dcp-comm-backend: "a2a" - max-num-seqs: 2 - max-num-batched-tokens: 8192 - trust-remote-code: true - max-cudagraph-capture-size: 1024 - stream-interval: 10 - language-model-only: true - moe-backend: "auto" - enable-flashinfer-autotune: true - enable-cumem-allocator: true - attention-backend: "TOKENSPEED_MLA" - attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":7,"draft_sample_method":"greedy","rejection_sample_method":"block"}' - enable-prefix-caching: true - prefix-match-unit: 128 - -sbatch_directives: - mem: "0" - cpus-per-task: "72" - comment: >- - '{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will - take extra loading time."}}' - -srun_options: - mem: "0" - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 - -benchmark: - type: custom - concurrencies: [1, 4] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - ENABLE_AGENTX_POWER: "1" - REQUIRE_POWER: "1" - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" - placement: - node: head diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p1d-dcp8-dcp8-dspark4-mooncake.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p1d-dcp8-dcp8-dspark4-mooncake.yaml deleted file mode 100644 index dd9a30f069..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p1d-dcp8-dcp8-dspark4-mooncake.yaml +++ /dev/null @@ -1,205 +0,0 @@ -schema: 2 -name: "kimi-k3-vllm-disagg-gb300-1p1d-dcp8-dcp8-dspark4-mooncake-agentic" - -model: - path: "moonshotai/Kimi-K3" - container: "vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56" - frameworks: - dynamo: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" - -dynamo: - install: true - - source: - rev: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" -slurm: - time_limit: "04:00:00" - -health_check: - max_attempts: 720 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - args: - - "--default_kv_lease_ttl=60000" - - "--eviction_high_watermark_ratio=0.95" - - "--eviction_ratio=0.10" - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "160GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - env: &kimi_env - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: "0" - DYN_REQUEST_PLANE: "tcp" - ETCD_LEASE_TTL: "600" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - TILELANG_CLEANUP_TEMP_FILES: "1" - VLLM_USE_NCCL_SYMM_MEM: "0" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_USE_RUST_FRONTEND: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - MC_SLICE_SIZE: "1048576" - MC_MAX_MR_SIZE: "4294967296" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_CONNECTOR_PREFETCH_KV_CAP: "0.65" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - NCCL_P2P_LEVEL: "NVL" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - WITH_NVIDIA_PEERMEM: "0" - NCCL_NET_PLUGIN: "none" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_RCACHE_MAX_UNRELEASED: "1024" - UCX_TCP_AF_PRIO: "inet" - VLLM_SSM_CONV_STATE_LAYOUT: "DS" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "moonshotai/Kimi-K3" - prefix-match-unit: 128 - load-format: "fastsafetensors" - kv-cache-dtype: "fp8" - mamba-ssm-cache-dtype: "bfloat16" - gpu-memory-utilization: 0.92 - tensor-parallel-size: 8 - decode-context-parallel-size: 8 - cp-kv-cache-interleave-size: 1 - dcp-comm-backend: "a2a" - enable-flashinfer-autotune: true - enable-cumem-allocator: true - trust-remote-code: true - max-cudagraph-capture-size: 512 - stream-interval: 10 - language-model-only: true - attention-backend: "TOKENSPEED_MLA" - attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' - enable-prefix-caching: true - kv_events: true - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: - <<: *kimi_env - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "moonshotai/Kimi-K3" - prefix-match-unit: 128 - load-format: "fastsafetensors" - kv-cache-dtype: "fp8" - mamba-ssm-cache-dtype: "bfloat16" - gpu-memory-utilization: 0.92 - tensor-parallel-size: 8 - decode-context-parallel-size: 8 - cp-kv-cache-interleave-size: 1 - dcp-comm-backend: "a2a" - enable-flashinfer-autotune: true - enable-cumem-allocator: true - trust-remote-code: true - max-cudagraph-capture-size: 512 - stream-interval: 10 - language-model-only: true - attention-backend: "TOKENSPEED_MLA" - attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' - enable-prefix-caching: true - -sbatch_directives: - mem: "0" - cpus-per-task: "72" - comment: >- - '{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will - take extra loading time."}}' - -srun_options: - mem: "0" - container-remap-root: "" - -telemetry: - enabled: true - provider: dcgm-power - default_frequency: 1.0 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 - -benchmark: - type: custom - client_placement: head - concurrencies: [48, 52, 56] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - ENABLE_AGENTX_POWER: "1" - REQUIRE_POWER: "1" - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p2d-dcp8-dcp8-dspark4-mooncake.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p2d-dcp8-dcp8-dspark4-mooncake.yaml deleted file mode 100644 index 6ea49208c6..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p2d-dcp8-dcp8-dspark4-mooncake.yaml +++ /dev/null @@ -1,205 +0,0 @@ -schema: 2 -name: "kimi-k3-vllm-disagg-gb300-1p2d-dcp8-dcp8-dspark4-mooncake-agentic" - -model: - path: "moonshotai/Kimi-K3" - container: "vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56" - frameworks: - dynamo: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" - -dynamo: - install: true - - source: - rev: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" -slurm: - time_limit: "04:00:00" - -health_check: - max_attempts: 720 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - args: - - "--default_kv_lease_ttl=60000" - - "--eviction_high_watermark_ratio=0.95" - - "--eviction_ratio=0.10" - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "160GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - env: &kimi_env - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: "0" - DYN_REQUEST_PLANE: "tcp" - ETCD_LEASE_TTL: "600" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - TILELANG_CLEANUP_TEMP_FILES: "1" - VLLM_USE_NCCL_SYMM_MEM: "0" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_USE_RUST_FRONTEND: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - MC_SLICE_SIZE: "1048576" - MC_MAX_MR_SIZE: "4294967296" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_CONNECTOR_PREFETCH_KV_CAP: "0.65" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - NCCL_P2P_LEVEL: "NVL" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - WITH_NVIDIA_PEERMEM: "0" - NCCL_NET_PLUGIN: "none" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_RCACHE_MAX_UNRELEASED: "1024" - UCX_TCP_AF_PRIO: "inet" - VLLM_SSM_CONV_STATE_LAYOUT: "DS" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "moonshotai/Kimi-K3" - prefix-match-unit: 128 - load-format: "fastsafetensors" - kv-cache-dtype: "fp8" - mamba-ssm-cache-dtype: "bfloat16" - gpu-memory-utilization: 0.92 - tensor-parallel-size: 8 - decode-context-parallel-size: 8 - cp-kv-cache-interleave-size: 1 - dcp-comm-backend: "a2a" - enable-flashinfer-autotune: true - enable-cumem-allocator: true - trust-remote-code: true - max-cudagraph-capture-size: 512 - stream-interval: 10 - language-model-only: true - attention-backend: "TOKENSPEED_MLA" - attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' - enable-prefix-caching: true - kv_events: true - decode: - nodes: 4 - workers: 2 - gpus: 8 - - env: - <<: *kimi_env - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "moonshotai/Kimi-K3" - prefix-match-unit: 128 - load-format: "fastsafetensors" - kv-cache-dtype: "fp8" - mamba-ssm-cache-dtype: "bfloat16" - gpu-memory-utilization: 0.92 - tensor-parallel-size: 8 - decode-context-parallel-size: 8 - cp-kv-cache-interleave-size: 1 - dcp-comm-backend: "a2a" - enable-flashinfer-autotune: true - enable-cumem-allocator: true - trust-remote-code: true - max-cudagraph-capture-size: 512 - stream-interval: 10 - language-model-only: true - attention-backend: "TOKENSPEED_MLA" - attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' - enable-prefix-caching: true - -sbatch_directives: - mem: "0" - cpus-per-task: "72" - comment: >- - '{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will - take extra loading time."}}' - -srun_options: - mem: "0" - container-remap-root: "" - -telemetry: - enabled: true - provider: dcgm-power - default_frequency: 1.0 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 - -benchmark: - type: custom - client_placement: head - concurrencies: [32, 48, 64] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - ENABLE_AGENTX_POWER: "1" - REQUIRE_POWER: "1" - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p3d-dcp8-dcp8-dspark4-mooncake.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p3d-dcp8-dcp8-dspark4-mooncake.yaml deleted file mode 100644 index 4bf4ecc656..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p3d-dcp8-dcp8-dspark4-mooncake.yaml +++ /dev/null @@ -1,205 +0,0 @@ -schema: 2 -name: "kimi-k3-vllm-disagg-gb300-1p3d-dcp8-dcp8-dspark4-mooncake-agentic" - -model: - path: "moonshotai/Kimi-K3" - container: "vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56" - frameworks: - dynamo: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" - -dynamo: - install: true - - source: - rev: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" -slurm: - time_limit: "04:00:00" - -health_check: - max_attempts: 720 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - args: - - "--default_kv_lease_ttl=60000" - - "--eviction_high_watermark_ratio=0.95" - - "--eviction_ratio=0.10" - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "160GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - env: &kimi_env - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: "0" - DYN_REQUEST_PLANE: "tcp" - ETCD_LEASE_TTL: "600" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - TILELANG_CLEANUP_TEMP_FILES: "1" - VLLM_USE_NCCL_SYMM_MEM: "0" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_USE_RUST_FRONTEND: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - MC_SLICE_SIZE: "1048576" - MC_MAX_MR_SIZE: "4294967296" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_CONNECTOR_PREFETCH_KV_CAP: "0.65" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - NCCL_P2P_LEVEL: "NVL" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - WITH_NVIDIA_PEERMEM: "0" - NCCL_NET_PLUGIN: "none" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_RCACHE_MAX_UNRELEASED: "1024" - UCX_TCP_AF_PRIO: "inet" - VLLM_SSM_CONV_STATE_LAYOUT: "DS" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "moonshotai/Kimi-K3" - prefix-match-unit: 128 - load-format: "fastsafetensors" - kv-cache-dtype: "fp8" - mamba-ssm-cache-dtype: "bfloat16" - gpu-memory-utilization: 0.92 - tensor-parallel-size: 8 - decode-context-parallel-size: 8 - cp-kv-cache-interleave-size: 1 - dcp-comm-backend: "a2a" - enable-flashinfer-autotune: true - enable-cumem-allocator: true - trust-remote-code: true - max-cudagraph-capture-size: 512 - stream-interval: 10 - language-model-only: true - attention-backend: "TOKENSPEED_MLA" - attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' - enable-prefix-caching: true - kv_events: true - decode: - nodes: 6 - workers: 3 - gpus: 8 - - env: - <<: *kimi_env - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "moonshotai/Kimi-K3" - prefix-match-unit: 128 - load-format: "fastsafetensors" - kv-cache-dtype: "fp8" - mamba-ssm-cache-dtype: "bfloat16" - gpu-memory-utilization: 0.92 - tensor-parallel-size: 8 - decode-context-parallel-size: 8 - cp-kv-cache-interleave-size: 1 - dcp-comm-backend: "a2a" - enable-flashinfer-autotune: true - enable-cumem-allocator: true - trust-remote-code: true - max-cudagraph-capture-size: 512 - stream-interval: 10 - language-model-only: true - attention-backend: "TOKENSPEED_MLA" - attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' - enable-prefix-caching: true - -sbatch_directives: - mem: "0" - cpus-per-task: "72" - comment: >- - '{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will - take extra loading time."}}' - -srun_options: - mem: "0" - container-remap-root: "" - -telemetry: - enabled: true - provider: dcgm-power - default_frequency: 1.0 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 - -benchmark: - type: custom - client_placement: head - concurrencies: [32, 48] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - ENABLE_AGENTX_POWER: "1" - REQUIRE_POWER: "1" - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p3d-dcp8-dcp8-dspark7-mooncake.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p3d-dcp8-dcp8-dspark7-mooncake.yaml deleted file mode 100644 index 0cd2a16883..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p3d-dcp8-dcp8-dspark7-mooncake.yaml +++ /dev/null @@ -1,205 +0,0 @@ -schema: 2 -name: "kimi-k3-vllm-disagg-gb300-1p3d-dcp8-dcp8-dspark7-mooncake-agentic" - -model: - path: "moonshotai/Kimi-K3" - container: "vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - container: - image: "vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56" - frameworks: - dynamo: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" - -dynamo: - install: true - - source: - rev: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" -slurm: - time_limit: "04:00:00" - -health_check: - max_attempts: 720 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - args: - - "--default_kv_lease_ttl=60000" - - "--eviction_high_watermark_ratio=0.95" - - "--eviction_ratio=0.10" - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "160GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: false -frontend: - type: dynamo - enable_multiple_frontends: false - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 900 - env: - DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - env: &kimi_env - VLLM_USE_DIRECT_DCP_A2A: "1" - VLLM_USE_DIRECT_DCP_Q_GATHER: "1" - VLLM_USE_DIRECT_DCP_KV_GATHER: "1" - VLLM_ALLREDUCE_USE_FLASHINFER: "1" - VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: "0" - DYN_REQUEST_PLANE: "tcp" - ETCD_LEASE_TTL: "600" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_RPC_TIMEOUT: "600000" - TILELANG_CLEANUP_TEMP_FILES: "1" - VLLM_USE_NCCL_SYMM_MEM: "0" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - VLLM_SERVER_DEV_MODE: "1" - VLLM_USE_V2_MODEL_RUNNER: "1" - VLLM_USE_RUST_FRONTEND: "1" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" - MC_SLICE_SIZE: "1048576" - MC_MAX_MR_SIZE: "4294967296" - VLLM_CONNECTOR_PREFETCH_DEPTH: "8" - VLLM_CONNECTOR_PREFETCH_KV_CAP: "0.65" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" - NCCL_P2P_LEVEL: "NVL" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - WITH_NVIDIA_PEERMEM: "0" - NCCL_NET_PLUGIN: "none" - UCX_MEMTYPE_CACHE: "n" - UCX_MEMTYPE_REG_WHOLE: "n" - UCX_RCACHE_MAX_UNRELEASED: "1024" - UCX_TCP_AF_PRIO: "inet" - VLLM_SSM_CONV_STATE_LAYOUT: "DS" - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "moonshotai/Kimi-K3" - prefix-match-unit: 128 - load-format: "fastsafetensors" - kv-cache-dtype: "fp8" - mamba-ssm-cache-dtype: "bfloat16" - gpu-memory-utilization: 0.92 - tensor-parallel-size: 8 - decode-context-parallel-size: 8 - cp-kv-cache-interleave-size: 1 - dcp-comm-backend: "a2a" - enable-flashinfer-autotune: true - enable-cumem-allocator: true - trust-remote-code: true - max-cudagraph-capture-size: 512 - stream-interval: 10 - language-model-only: true - attention-backend: "TOKENSPEED_MLA" - attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":7,"draft_sample_method":"greedy","rejection_sample_method":"block"}' - enable-prefix-caching: true - kv_events: true - decode: - nodes: 6 - workers: 3 - gpus: 8 - - env: - <<: *kimi_env - args: - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' - served-model-name: "moonshotai/Kimi-K3" - prefix-match-unit: 128 - load-format: "fastsafetensors" - kv-cache-dtype: "fp8" - mamba-ssm-cache-dtype: "bfloat16" - gpu-memory-utilization: 0.92 - tensor-parallel-size: 8 - decode-context-parallel-size: 8 - cp-kv-cache-interleave-size: 1 - dcp-comm-backend: "a2a" - enable-flashinfer-autotune: true - enable-cumem-allocator: true - trust-remote-code: true - max-cudagraph-capture-size: 512 - stream-interval: 10 - language-model-only: true - attention-backend: "TOKENSPEED_MLA" - attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' - speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":7,"draft_sample_method":"greedy","rejection_sample_method":"block"}' - enable-prefix-caching: true - -sbatch_directives: - mem: "0" - cpus-per-task: "72" - comment: >- - '{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will - take extra loading time."}}' - -srun_options: - mem: "0" - container-remap-root: "" - -telemetry: - enabled: true - provider: dcgm-power - default_frequency: 1.0 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 - -benchmark: - type: custom - client_placement: head - concurrencies: [1] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - ENABLE_AGENTX_POWER: "1" - REQUIRE_POWER: "1" - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..ab659b60ec --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml @@ -0,0 +1,762 @@ +# srt-slurm recipes for kimik3/vllm/gb300-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: moonshotai/Kimi-K3 + container: vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56 + precision: fp4 + identity: + model: + repo: moonshotai/Kimi-K3 + container: + image: vllm/vllm-openai:nightly-dc36fcce902a63eab06c1b93a5c4a5ee178a0c56 + frameworks: + dynamo: ba83080ecd31c1ce918559e576d3c5bc9e092ff1 + dynamo: + install: true + source: + rev: ba83080ecd31c1ce918559e576d3c5bc9e092ff1 + slurm: + time_limit: 04:00:00 + health_check: + max_attempts: 720 + interval_seconds: 10 + resources: + gpu_type: gb300 + gpus_per_node: 4 + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + args: + - --default_kv_lease_ttl=60000 + - --eviction_high_watermark_ratio=0.95 + - --eviction_ratio=0.10 + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 160GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: false + frontend: + type: dynamo + enable_multiple_frontends: false + args: + router-mode: least-loaded + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600' + DYN_TOKENIZER_CACHE_BYTES: '8589934592' + engine: + type: vllm + connector: null + roles: {} + sbatch_directives: + mem: '0' + cpus-per-task: '72' + comment: '''{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will take extra loading time."}}''' + srun_options: + mem: '0' + container-remap-root: '' + telemetry: + enabled: true + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + collector_join_timeout_seconds: 12 + dcgm_exporter: + container_image: dcgm-exporter + port: 19401 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + ENABLE_AGENTX_POWER: '1' + REQUIRE_POWER: '1' + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + +override_agg_dcp8_dspark4_mooncake: + name: kimi-k3-vllm-agg-gb300-dcp8-dspark4-mooncake-agentic + roles: + agg: + nodes: 2 + workers: 1 + gpus: 8 + env: + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + # Use the direct DCP a2a/gather kernels rather than the collective + # fallback, matching the B300 arm (kimik3_fp4_b300_vllm_mtp.sh). + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + ETCD_LEASE_TTL: '120' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + PYTHONNOUSERSITE: '1' + TORCH_CUDA_ARCH_LIST: '10.0' + PYTHONHASHSEED: '42' + VLLM_HTTP_TIMEOUT_KEEP_ALIVE: '900' + MC_SLICE_SIZE: '1048576' + MC_MAX_MR_SIZE: '4294967296' + WITH_NVIDIA_PEERMEM: '0' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + args: + served-model-name: moonshotai/Kimi-K3 + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + dcp-comm-backend: a2a + gpu-memory-utilization: 0.92 + max-num-batched-tokens: 16384 + trust-remote-code: true + language-model-only: true + load-format: fastsafetensors + moe-backend: auto + enable-flashinfer-autotune: true + enable-prefix-caching: true + prefix-match-unit: 128 + kv-cache-dtype: fp8 + mamba-ssm-cache-dtype: bfloat16 + stream-interval: 10 + attention-backend: TOKENSPEED_MLA + attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' + max-cudagraph-capture-size: 1024 + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_offload":false}}' + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' + telemetry: + collect_interval_ms: 1000 + benchmark: + concurrencies: [48, 52, 56] + env: + AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: '300' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' + AIPERF_HTTP_TCP_USER_TIMEOUT: '900000' + placement: + node: head + +override_agg_dcp8_dspark7_maxseq2_mooncake: + name: kimi-k3-vllm-agg-gb300-dcp8-dspark7-maxseq2-mooncake-agentic + roles: + agg: + nodes: 2 + workers: 1 + gpus: 8 + env: + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1' + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + ETCD_LEASE_TTL: '120' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + MC_SLICE_SIZE: '1048576' + MC_MAX_MR_SIZE: '4294967296' + WITH_NVIDIA_PEERMEM: '0' + VLLM_LOG_STATS_INTERVAL: '1' + args: + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_offload":false}}' + served-model-name: moonshotai/Kimi-K3 + load-format: fastsafetensors + kv-cache-dtype: fp8 + mamba-ssm-cache-dtype: bfloat16 + gpu-memory-utilization: 0.92 + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + dcp-comm-backend: a2a + max-num-seqs: 2 + max-num-batched-tokens: 8192 + trust-remote-code: true + max-cudagraph-capture-size: 1024 + stream-interval: 10 + language-model-only: true + moe-backend: auto + enable-flashinfer-autotune: true + enable-cumem-allocator: true + attention-backend: TOKENSPEED_MLA + attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":7,"draft_sample_method":"greedy","rejection_sample_method":"block"}' + enable-prefix-caching: true + prefix-match-unit: 128 + telemetry: + collect_interval_ms: 1000 + benchmark: + concurrencies: [1, 4] + placement: + node: head + +override_disagg_1p1d_dcp8_dcp8_dspark4_mooncake: + name: kimi-k3-vllm-disagg-gb300-1p1d-dcp8-dcp8-dspark4-mooncake-agentic + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + env: + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0' + DYN_REQUEST_PLANE: tcp + ETCD_LEASE_TTL: '600' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + MC_SLICE_SIZE: '1048576' + MC_MAX_MR_SIZE: '4294967296' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + NCCL_P2P_LEVEL: NVL + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + WITH_NVIDIA_PEERMEM: '0' + NCCL_NET_PLUGIN: none + UCX_MEMTYPE_CACHE: n + UCX_MEMTYPE_REG_WHOLE: n + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_TCP_AF_PRIO: inet + VLLM_SSM_CONV_STATE_LAYOUT: DS + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: moonshotai/Kimi-K3 + prefix-match-unit: 128 + load-format: fastsafetensors + kv-cache-dtype: fp8 + mamba-ssm-cache-dtype: bfloat16 + gpu-memory-utilization: 0.92 + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + cp-kv-cache-interleave-size: 1 + dcp-comm-backend: a2a + enable-flashinfer-autotune: true + enable-cumem-allocator: true + trust-remote-code: true + max-cudagraph-capture-size: 512 + stream-interval: 10 + language-model-only: true + attention-backend: TOKENSPEED_MLA + attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' + enable-prefix-caching: true + kv_events: true + decode: + nodes: 2 + workers: 1 + gpus: 8 + env: + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0' + DYN_REQUEST_PLANE: tcp + ETCD_LEASE_TTL: '600' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + MC_SLICE_SIZE: '1048576' + MC_MAX_MR_SIZE: '4294967296' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + NCCL_P2P_LEVEL: NVL + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + WITH_NVIDIA_PEERMEM: '0' + NCCL_NET_PLUGIN: none + UCX_MEMTYPE_CACHE: n + UCX_MEMTYPE_REG_WHOLE: n + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_TCP_AF_PRIO: inet + VLLM_SSM_CONV_STATE_LAYOUT: DS + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: moonshotai/Kimi-K3 + prefix-match-unit: 128 + load-format: fastsafetensors + kv-cache-dtype: fp8 + mamba-ssm-cache-dtype: bfloat16 + gpu-memory-utilization: 0.92 + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + cp-kv-cache-interleave-size: 1 + dcp-comm-backend: a2a + enable-flashinfer-autotune: true + enable-cumem-allocator: true + trust-remote-code: true + max-cudagraph-capture-size: 512 + stream-interval: 10 + language-model-only: true + attention-backend: TOKENSPEED_MLA + attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' + enable-prefix-caching: true + telemetry: + provider: dcgm-power + default_frequency: 1.0 + benchmark: + concurrencies: [48, 52, 56] + env: + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + client_placement: head + +override_disagg_1p2d_dcp8_dcp8_dspark4_mooncake: + name: kimi-k3-vllm-disagg-gb300-1p2d-dcp8-dcp8-dspark4-mooncake-agentic + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + env: + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0' + DYN_REQUEST_PLANE: tcp + ETCD_LEASE_TTL: '600' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + MC_SLICE_SIZE: '1048576' + MC_MAX_MR_SIZE: '4294967296' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + NCCL_P2P_LEVEL: NVL + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + WITH_NVIDIA_PEERMEM: '0' + NCCL_NET_PLUGIN: none + UCX_MEMTYPE_CACHE: n + UCX_MEMTYPE_REG_WHOLE: n + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_TCP_AF_PRIO: inet + VLLM_SSM_CONV_STATE_LAYOUT: DS + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: moonshotai/Kimi-K3 + prefix-match-unit: 128 + load-format: fastsafetensors + kv-cache-dtype: fp8 + mamba-ssm-cache-dtype: bfloat16 + gpu-memory-utilization: 0.92 + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + cp-kv-cache-interleave-size: 1 + dcp-comm-backend: a2a + enable-flashinfer-autotune: true + enable-cumem-allocator: true + trust-remote-code: true + max-cudagraph-capture-size: 512 + stream-interval: 10 + language-model-only: true + attention-backend: TOKENSPEED_MLA + attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' + enable-prefix-caching: true + kv_events: true + decode: + nodes: 4 + workers: 2 + gpus: 8 + env: + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0' + DYN_REQUEST_PLANE: tcp + ETCD_LEASE_TTL: '600' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + MC_SLICE_SIZE: '1048576' + MC_MAX_MR_SIZE: '4294967296' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + NCCL_P2P_LEVEL: NVL + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + WITH_NVIDIA_PEERMEM: '0' + NCCL_NET_PLUGIN: none + UCX_MEMTYPE_CACHE: n + UCX_MEMTYPE_REG_WHOLE: n + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_TCP_AF_PRIO: inet + VLLM_SSM_CONV_STATE_LAYOUT: DS + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: moonshotai/Kimi-K3 + prefix-match-unit: 128 + load-format: fastsafetensors + kv-cache-dtype: fp8 + mamba-ssm-cache-dtype: bfloat16 + gpu-memory-utilization: 0.92 + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + cp-kv-cache-interleave-size: 1 + dcp-comm-backend: a2a + enable-flashinfer-autotune: true + enable-cumem-allocator: true + trust-remote-code: true + max-cudagraph-capture-size: 512 + stream-interval: 10 + language-model-only: true + attention-backend: TOKENSPEED_MLA + attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' + enable-prefix-caching: true + telemetry: + provider: dcgm-power + default_frequency: 1.0 + benchmark: + concurrencies: [32, 48, 64] + env: + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + client_placement: head + +override_disagg_1p3d_dcp8_dcp8_dspark4_mooncake: + name: kimi-k3-vllm-disagg-gb300-1p3d-dcp8-dcp8-dspark4-mooncake-agentic + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + env: + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0' + DYN_REQUEST_PLANE: tcp + ETCD_LEASE_TTL: '600' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + MC_SLICE_SIZE: '1048576' + MC_MAX_MR_SIZE: '4294967296' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + NCCL_P2P_LEVEL: NVL + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + WITH_NVIDIA_PEERMEM: '0' + NCCL_NET_PLUGIN: none + UCX_MEMTYPE_CACHE: n + UCX_MEMTYPE_REG_WHOLE: n + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_TCP_AF_PRIO: inet + VLLM_SSM_CONV_STATE_LAYOUT: DS + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: moonshotai/Kimi-K3 + prefix-match-unit: 128 + load-format: fastsafetensors + kv-cache-dtype: fp8 + mamba-ssm-cache-dtype: bfloat16 + gpu-memory-utilization: 0.92 + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + cp-kv-cache-interleave-size: 1 + dcp-comm-backend: a2a + enable-flashinfer-autotune: true + enable-cumem-allocator: true + trust-remote-code: true + max-cudagraph-capture-size: 512 + stream-interval: 10 + language-model-only: true + attention-backend: TOKENSPEED_MLA + attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' + enable-prefix-caching: true + kv_events: true + decode: + nodes: 6 + workers: 3 + gpus: 8 + env: + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0' + DYN_REQUEST_PLANE: tcp + ETCD_LEASE_TTL: '600' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + MC_SLICE_SIZE: '1048576' + MC_MAX_MR_SIZE: '4294967296' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + NCCL_P2P_LEVEL: NVL + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + WITH_NVIDIA_PEERMEM: '0' + NCCL_NET_PLUGIN: none + UCX_MEMTYPE_CACHE: n + UCX_MEMTYPE_REG_WHOLE: n + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_TCP_AF_PRIO: inet + VLLM_SSM_CONV_STATE_LAYOUT: DS + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: moonshotai/Kimi-K3 + prefix-match-unit: 128 + load-format: fastsafetensors + kv-cache-dtype: fp8 + mamba-ssm-cache-dtype: bfloat16 + gpu-memory-utilization: 0.92 + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + cp-kv-cache-interleave-size: 1 + dcp-comm-backend: a2a + enable-flashinfer-autotune: true + enable-cumem-allocator: true + trust-remote-code: true + max-cudagraph-capture-size: 512 + stream-interval: 10 + language-model-only: true + attention-backend: TOKENSPEED_MLA + attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy","rejection_sample_method":"block"}' + enable-prefix-caching: true + telemetry: + provider: dcgm-power + default_frequency: 1.0 + benchmark: + concurrencies: [32, 48] + env: + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + client_placement: head + +override_disagg_1p3d_dcp8_dcp8_dspark7_mooncake: + name: kimi-k3-vllm-disagg-gb300-1p3d-dcp8-dcp8-dspark7-mooncake-agentic + engine: + dp_launch_mode: per_node + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + env: + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0' + DYN_REQUEST_PLANE: tcp + ETCD_LEASE_TTL: '600' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + MC_SLICE_SIZE: '1048576' + MC_MAX_MR_SIZE: '4294967296' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + NCCL_P2P_LEVEL: NVL + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + WITH_NVIDIA_PEERMEM: '0' + NCCL_NET_PLUGIN: none + UCX_MEMTYPE_CACHE: n + UCX_MEMTYPE_REG_WHOLE: n + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_TCP_AF_PRIO: inet + VLLM_SSM_CONV_STATE_LAYOUT: DS + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: moonshotai/Kimi-K3 + prefix-match-unit: 128 + load-format: fastsafetensors + kv-cache-dtype: fp8 + mamba-ssm-cache-dtype: bfloat16 + gpu-memory-utilization: 0.92 + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + cp-kv-cache-interleave-size: 1 + dcp-comm-backend: a2a + enable-flashinfer-autotune: true + enable-cumem-allocator: true + trust-remote-code: true + max-cudagraph-capture-size: 512 + stream-interval: 10 + language-model-only: true + attention-backend: TOKENSPEED_MLA + attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":7,"draft_sample_method":"greedy","rejection_sample_method":"block"}' + enable-prefix-caching: true + kv_events: true + decode: + nodes: 6 + workers: 3 + gpus: 8 + env: + VLLM_USE_DIRECT_DCP_A2A: '1' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0' + DYN_REQUEST_PLANE: tcp + ETCD_LEASE_TTL: '600' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_RPC_TIMEOUT: '600000' + TILELANG_CLEANUP_TEMP_FILES: '1' + VLLM_USE_NCCL_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + VLLM_SERVER_DEV_MODE: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' + MC_SLICE_SIZE: '1048576' + MC_MAX_MR_SIZE: '4294967296' + VLLM_CONNECTOR_PREFETCH_DEPTH: '8' + VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' + NCCL_P2P_LEVEL: NVL + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + WITH_NVIDIA_PEERMEM: '0' + NCCL_NET_PLUGIN: none + UCX_MEMTYPE_CACHE: n + UCX_MEMTYPE_REG_WHOLE: n + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_TCP_AF_PRIO: inet + VLLM_SSM_CONV_STATE_LAYOUT: DS + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: moonshotai/Kimi-K3 + prefix-match-unit: 128 + load-format: fastsafetensors + kv-cache-dtype: fp8 + mamba-ssm-cache-dtype: bfloat16 + gpu-memory-utilization: 0.92 + tensor-parallel-size: 8 + decode-context-parallel-size: 8 + cp-kv-cache-interleave-size: 1 + dcp-comm-backend: a2a + enable-flashinfer-autotune: true + enable-cumem-allocator: true + trust-remote-code: true + max-cudagraph-capture-size: 512 + stream-interval: 10 + language-model-only: true + attention-backend: TOKENSPEED_MLA + attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}' + speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":7,"draft_sample_method":"greedy","rejection_sample_method":"block"}' + enable-prefix-caching: true + telemetry: + provider: dcgm-power + default_frequency: 1.0 + benchmark: + concurrencies: [1] + env: + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + client_placement: head diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/agg-tp16dp2ep32-latency.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/agg-tp16dp2ep32-latency.yaml deleted file mode 100644 index 0ae9ca0fe6..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/agg-tp16dp2ep32-latency.yaml +++ /dev/null @@ -1,109 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-h200-tp16dp2ep32-latency-agentic" - -model: - path: "kimik3" - container: "vllm/vllm-openai:kimi-k3" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - revision: "9f62e4e9fffbd0a83ddd60e1c209d828994b3569" - container: - image: "vllm/vllm-openai:kimi-k3" - frameworks: - vllm: "0.1.dev19262+gb6bbf29dd.d20260727" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 720 - interval_seconds: 10 - -resources: - gpu_type: "h200" - gpus_per_node: 8 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 32 - - env: - CUDA_LAUNCH_BLOCKING: "1" - GLOO_SOCKET_IFNAME: "eth0" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - NCCL_SOCKET_IFNAME: "eth0" - NCCL_CUMEM_ENABLE: "1" - PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" - PYTHONNOUSERSITE: "1" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_ROUTED_DOWN_PROJ_STREAM_TOKEN_THRESHOLD: "0" - VLLM_USE_V2_MODEL_RUNNER: "1" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 16 - data-parallel-size: 2 - enable-expert-parallel: true - trust-remote-code: true - load-format: fastsafetensors - moe-backend: marlin - attention-backend: FLASHMLA - gpu-memory-utilization: 0.975 - max-num-seqs: 5 - max-num-batched-tokens: 4096 - speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - enforce-eager: true - compilation-config: '{"pass_config":{"fuse_allreduce_rms":false}}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - no-enable-flashinfer-autotune: true - disable-custom-all-reduce: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - concurrencies: [1, 2, 3, 4, 5, 6, 7, 8, 10, 12] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/agg-tp8dp4ep32-balanced.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/agg-tp8dp4ep32-balanced.yaml deleted file mode 100644 index 6c241ec561..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/agg-tp8dp4ep32-balanced.yaml +++ /dev/null @@ -1,111 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-h200-tp8dp4ep32-balanced-agentic" - -model: - path: "kimik3" - container: "vllm/vllm-openai:kimi-k3" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - revision: "9f62e4e9fffbd0a83ddd60e1c209d828994b3569" - container: - image: "vllm/vllm-openai:kimi-k3" - frameworks: - vllm: "0.1.dev19262+gb6bbf29dd.d20260727" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 720 - interval_seconds: 10 - -resources: - gpu_type: "h200" - gpus_per_node: 8 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 32 - - env: - CUDA_LAUNCH_BLOCKING: "1" - GLOO_SOCKET_IFNAME: "eth0" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - NCCL_SOCKET_IFNAME: "eth0" - NCCL_CUMEM_ENABLE: "1" - PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" - PYTHONNOUSERSITE: "1" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_ROUTED_DOWN_PROJ_STREAM_TOKEN_THRESHOLD: "0" - VLLM_USE_V2_MODEL_RUNNER: "1" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - data-parallel-size: 4 - enable-expert-parallel: true - trust-remote-code: true - load-format: fastsafetensors - moe-backend: marlin - attention-backend: FLASHMLA - gpu-memory-utilization: 0.95 - max-num-seqs: 8 - max-num-batched-tokens: 4096 - speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - enforce-eager: true - no-async-scheduling: true - compilation-config: '{"pass_config":{"fuse_allreduce_rms":false}}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - no-enable-flashinfer-autotune: true - disable-custom-all-reduce: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - provider: dcgm-power - default_frequency: 1.0 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - concurrencies: [1, 2, 3, 4, 5, 6, 7, 8, 10, 12, 14, 16] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/agg-tp8dp4ep32-vllm-simple.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/agg-tp8dp4ep32-vllm-simple.yaml deleted file mode 100644 index edfb3d9963..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/agg-tp8dp4ep32-vllm-simple.yaml +++ /dev/null @@ -1,112 +0,0 @@ -schema: 2 -name: "kimik3-vllm-agg-h200-tp8dp4ep32-vllm-simple-agentic" - -model: - path: "kimik3" - container: "vllm/vllm-openai:kimi-k3" - precision: "fp4" - -identity: - model: - repo: "moonshotai/Kimi-K3" - revision: "9f62e4e9fffbd0a83ddd60e1c209d828994b3569" - container: - image: "vllm/vllm-openai:kimi-k3" - frameworks: - vllm: "0.1.dev19262+gb6bbf29dd.d20260727" - -dynamo: - install: false - -slurm: - time_limit: "8:00:00" - -health_check: - max_attempts: 720 - interval_seconds: 10 - -resources: - gpu_type: "h200" - gpus_per_node: 8 -frontend: - type: vllm - enable_multiple_frontends: false - -engine: - type: vllm - connector: -roles: - agg: - nodes: 4 - workers: 1 - gpus: 32 - - env: - CUDA_LAUNCH_BLOCKING: "1" - GLOO_SOCKET_IFNAME: "eth0" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - NCCL_SOCKET_IFNAME: "eth0" - NCCL_CUMEM_ENABLE: "1" - PYTHONHASHSEED: "42" - PYTHONNOUSERSITE: "1" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_ROUTED_DOWN_PROJ_STREAM_TOKEN_THRESHOLD: "0" - VLLM_USE_V2_MODEL_RUNNER: "1" - args: - served-model-name: "moonshotai/Kimi-K3" - tensor-parallel-size: 8 - data-parallel-size: 4 - enable-expert-parallel: true - trust-remote-code: true - load-format: fastsafetensors - moe-backend: marlin - attention-backend: FLASHMLA - gpu-memory-utilization: 0.95 - max-num-seqs: 16 - max-num-batched-tokens: 4096 - speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"draft_sample_method":"probabilistic","rejection_sample_method":"block"}' - enforce-eager: true - no-async-scheduling: true - compilation-config: '{"pass_config":{"fuse_allreduce_rms":false}}' - enable-prefix-caching: true - enable-prompt-tokens-details: true - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use_per_rank":154250000000,"lazy_offload":false}}' - no-enable-flashinfer-autotune: true - disable-custom-all-reduce: true - enable-auto-tool-choice: true - tool-call-parser: kimi_k3 - reasoning-parser: kimi_k3 - language-model-only: true - -srun_options: - container-remap-root: "" - -telemetry: - enabled: true - provider: dcgm-power - default_frequency: 1.0 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 - -benchmark: - type: custom - concurrencies: [8, 10, 12, 14, 16, 18, 20, 22, 24, 26, 28, 30, 32] - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..4d1325d9f2 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/variants.yaml @@ -0,0 +1,149 @@ +# srt-slurm recipes for kimik3/vllm/h200-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/kimik3/vllm/h200-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: kimik3 + container: vllm/vllm-openai:kimi-k3 + precision: fp4 + identity: + model: + repo: moonshotai/Kimi-K3 + revision: 9f62e4e9fffbd0a83ddd60e1c209d828994b3569 + container: + image: vllm/vllm-openai:kimi-k3 + frameworks: + vllm: 0.1.dev19262+gb6bbf29dd.d20260727 + dynamo: + install: false + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 720 + interval_seconds: 10 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + engine: + type: vllm + connector: null + roles: + agg: + nodes: 4 + workers: 1 + gpus: 32 + env: + CUDA_LAUNCH_BLOCKING: '1' + GLOO_SOCKET_IFNAME: eth0 + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + NCCL_SOCKET_IFNAME: eth0 + NCCL_CUMEM_ENABLE: '1' + PYTHONNOUSERSITE: '1' + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '7200' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_ROUTED_DOWN_PROJ_STREAM_TOKEN_THRESHOLD: '0' + VLLM_USE_V2_MODEL_RUNNER: '1' + args: + served-model-name: moonshotai/Kimi-K3 + enable-expert-parallel: true + trust-remote-code: true + load-format: fastsafetensors + moe-backend: marlin + attention-backend: FLASHMLA + max-num-batched-tokens: 4096 + speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + enforce-eager: true + compilation-config: '{"pass_config":{"fuse_allreduce_rms":false}}' + enable-prefix-caching: true + enable-prompt-tokens-details: true + no-enable-flashinfer-autotune: true + disable-custom-all-reduce: true + enable-auto-tool-choice: true + tool-call-parser: kimi_k3 + reasoning-parser: kimi_k3 + language-model-only: true + srun_options: + container-remap-root: '' + telemetry: + enabled: true + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + collector_join_timeout_seconds: 12 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: '300' + HF_HUB_CACHE: /hf_hub_cache + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + +override_agg_tp16dp2ep32_latency: + name: kimik3-vllm-agg-h200-tp16dp2ep32-latency-agentic + roles: + agg: + env: + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + args: + tensor-parallel-size: 16 + data-parallel-size: 2 + gpu-memory-utilization: 0.975 + max-num-seqs: 5 + telemetry: + collect_interval_ms: 1000 + benchmark: + concurrencies: [1, 2, 3, 4, 5, 6, 7, 8, 10, 12] + +override_agg_tp8dp4ep32_balanced: + name: kimik3-vllm-agg-h200-tp8dp4ep32-balanced-agentic + roles: + agg: + env: + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + args: + tensor-parallel-size: 8 + data-parallel-size: 4 + gpu-memory-utilization: 0.95 + max-num-seqs: 8 + no-async-scheduling: true + telemetry: + provider: dcgm-power + default_frequency: 1.0 + benchmark: + concurrencies: [1, 2, 3, 4, 5, 6, 7, 8, 10, 12, 14, 16] + +override_agg_tp8dp4ep32_vllm_simple: + name: kimik3-vllm-agg-h200-tp8dp4ep32-vllm-simple-agentic + roles: + agg: + env: + PYTHONHASHSEED: '42' + args: + tensor-parallel-size: 8 + data-parallel-size: 4 + gpu-memory-utilization: 0.95 + max-num-seqs: 16 + no-async-scheduling: true + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use_per_rank":154250000000,"lazy_offload":false}}' + telemetry: + provider: dcgm-power + default_frequency: 1.0 + benchmark: + concurrencies: [8, 10, 12, 14, 16, 18, 20, 22, 24, 26, 28, 30, 32] diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c10-b10-eagle3.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c10-b10-eagle3.yaml deleted file mode 100644 index c9574016ba..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c10-b10-eagle3.yaml +++ /dev/null @@ -1,152 +0,0 @@ -# MiniMax-M3 NVFP4, TensorRT-LLM aggregated TP4 + EAGLE3-GQA (3 draft tokens), one GB200 node, Dynamo frontend. -# Engine config is the max_batch_size=10 member of the TP4 recipe set. -# Host KV cache is set to 128 GiB per rank for the four-rank layout. -# AgentX acceptance is selected from the committed golden curve at submission. -# (= AL - 1 = 1.78 for the committed golden AL 2.78) for throughput runs and leaves eval-only runs on real acceptance. -schema: 2 -name: dynamo-agg-gb200-tp4-c10-b1-eagle3 -model: - path: minimax-m3-nvfp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc23.post1 - precision: fp4 -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - request_plane: tcp -health_check: - max_attempts: 270 - interval_seconds: 10 -resources: - gpu_type: gb200 - gpus_per_node: 4 -engine: - type: trtllm - numa_memory_bind: true - served_model_name: nvidia/MiniMax-M3-NVFP4 - publish_events_and_metrics: false -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - env: - HF_HUB_OFFLINE: '1' - TRANSFORMERS_OFFLINE: '1' - HF_HUB_CACHE: /hf_hub_cache - TQDM_DISABLE: '1' - HF_HUB_DISABLE_PROGRESS_BARS: '1' - TLLM_LOG_LEVEL: INFO - TLLM_PROFILE_LOG_RANKS: all - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - PYTHONNOUSERSITE: '1' - TRTLLM_SERVE_ENABLE_MSGSPEC: '1' - TRTLLM_TORCH_COMPILE_CONTEXT_ONLY: '1' - DYN_PUBLISH_KV_EVENTS: '0' - args: - max_seq_len: 1048576 - max_num_tokens: 16384 - max_batch_size: 10 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 3 - - 4 - - 5 - - 6 - - 7 - - 8 - - 9 - - 10 - torch_compile_config: - enable_fullgraph: true - enable_inductor: false - enable_piecewise_cuda_graph: true - capture_num_tokens: - - 1 - - 512 - - 1024 - - 2048 - enable_userbuffers: true - max_num_streams: 3 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - sparse_attention_config: - algorithm: minimax_m3 - implementation: msa - indexer_kv_dtype: fp8 - sparse_disable_index_value: true - fuse_qkv_index_projection: true - kv_cache_config: - free_gpu_memory_fraction: 0.94 - enable_block_reuse: true - block_reuse_policy: per_conversation - tokens_per_block: 128 - use_kv_cache_manager_v2: true - dtype: fp8 - event_buffer_max_size: 0 - host_cache_size: 137438953472 - speculative_config: - decoding_type: Eagle3 - max_draft_len: 3 - speculative_model: Inferact/MiniMax-M3-EAGLE3-GQA - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - enable_chunked_prefill: true - enable_autotuner: true - trust_remote_code: true - stream_interval: 20 - print_iter_log: true - num_postprocess_workers: 8 - enable_attention_dp: false - tensor_parallel_size: 4 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_ROUTER_TEMPERATURE: '0' - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TCP_REQUEST_TIMEOUT: '30' - DYN_LOG: warn - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: '0' - SERVED_MODEL_NAME: nvidia/MiniMax-M3-NVFP4 - MAX_MODEL_LEN: '1048576' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - OPENAI_API_KEY: EMPTY - KV_OFFLOADING: dram - KV_OFFLOAD_BACKEND: native - KV_OFFLOAD_BACKEND_METADATA: '{"name":"native"}' - TOTAL_CPU_DRAM_GB: '550' - MODEL: nvidia/MiniMax-M3-NVFP4 - MODEL_PREFIX: minimaxm3 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '10' - DURATION: '3600' diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c15-b15-eagle3.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c15-b15-eagle3.yaml deleted file mode 100644 index cef285921e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c15-b15-eagle3.yaml +++ /dev/null @@ -1,157 +0,0 @@ -# MiniMax-M3 NVFP4, TensorRT-LLM aggregated TP4 + EAGLE3-GQA (3 draft tokens), one GB200 node, Dynamo frontend. -# Engine config is the max_batch_size=15 member of the TP4 recipe set. -# Host KV cache is set to 128 GiB per rank for the four-rank layout. -# AgentX acceptance is selected from the committed golden curve at submission. -# (= AL - 1 = 1.78 for the committed golden AL 2.78) for throughput runs and leaves eval-only runs on real acceptance. -schema: 2 -name: dynamo-agg-gb200-tp4-c15-b1-eagle3 -model: - path: minimax-m3-nvfp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc23.post1 - precision: fp4 -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - request_plane: tcp -health_check: - max_attempts: 270 - interval_seconds: 10 -resources: - gpu_type: gb200 - gpus_per_node: 4 -engine: - type: trtllm - numa_memory_bind: true - served_model_name: nvidia/MiniMax-M3-NVFP4 - publish_events_and_metrics: false -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - env: - HF_HUB_OFFLINE: '1' - TRANSFORMERS_OFFLINE: '1' - HF_HUB_CACHE: /hf_hub_cache - TQDM_DISABLE: '1' - HF_HUB_DISABLE_PROGRESS_BARS: '1' - TLLM_LOG_LEVEL: INFO - TLLM_PROFILE_LOG_RANKS: all - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - PYTHONNOUSERSITE: '1' - TRTLLM_SERVE_ENABLE_MSGSPEC: '1' - TRTLLM_TORCH_COMPILE_CONTEXT_ONLY: '1' - DYN_PUBLISH_KV_EVENTS: '0' - args: - max_seq_len: 1048576 - max_num_tokens: 16384 - max_batch_size: 15 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 3 - - 4 - - 5 - - 6 - - 7 - - 8 - - 9 - - 10 - - 11 - - 12 - - 13 - - 14 - - 15 - torch_compile_config: - enable_fullgraph: true - enable_inductor: false - enable_piecewise_cuda_graph: true - capture_num_tokens: - - 1 - - 512 - - 1024 - - 2048 - enable_userbuffers: true - max_num_streams: 3 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - sparse_attention_config: - algorithm: minimax_m3 - implementation: msa - indexer_kv_dtype: fp8 - sparse_disable_index_value: true - fuse_qkv_index_projection: true - kv_cache_config: - free_gpu_memory_fraction: 0.94 - enable_block_reuse: true - block_reuse_policy: per_conversation - tokens_per_block: 128 - use_kv_cache_manager_v2: true - dtype: fp8 - event_buffer_max_size: 0 - host_cache_size: 137438953472 - speculative_config: - decoding_type: Eagle3 - max_draft_len: 3 - speculative_model: Inferact/MiniMax-M3-EAGLE3-GQA - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - enable_chunked_prefill: true - enable_autotuner: true - trust_remote_code: true - stream_interval: 20 - print_iter_log: true - num_postprocess_workers: 8 - enable_attention_dp: false - tensor_parallel_size: 4 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_ROUTER_TEMPERATURE: '0' - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TCP_REQUEST_TIMEOUT: '30' - DYN_LOG: warn - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: '0' - SERVED_MODEL_NAME: nvidia/MiniMax-M3-NVFP4 - MAX_MODEL_LEN: '1048576' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - OPENAI_API_KEY: EMPTY - KV_OFFLOADING: dram - KV_OFFLOAD_BACKEND: native - KV_OFFLOAD_BACKEND_METADATA: '{"name":"native"}' - TOTAL_CPU_DRAM_GB: '550' - MODEL: nvidia/MiniMax-M3-NVFP4 - MODEL_PREFIX: minimaxm3 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '15' - DURATION: '3600' diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c20-b20-eagle3.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c20-b20-eagle3.yaml deleted file mode 100644 index 55c9e38641..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c20-b20-eagle3.yaml +++ /dev/null @@ -1,162 +0,0 @@ -# MiniMax-M3 NVFP4, TensorRT-LLM aggregated TP4 + EAGLE3-GQA (3 draft tokens), one GB200 node, Dynamo frontend. -# Engine config is the max_batch_size=20 member of the TP4 recipe set. -# Host KV cache is set to 128 GiB per rank for the four-rank layout. -# AgentX acceptance is selected from the committed golden curve at submission. -# (= AL - 1 = 1.78 for the committed golden AL 2.78) for throughput runs and leaves eval-only runs on real acceptance. -schema: 2 -name: dynamo-agg-gb200-tp4-c20-b1-eagle3 -model: - path: minimax-m3-nvfp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc23.post1 - precision: fp4 -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - request_plane: tcp -health_check: - max_attempts: 270 - interval_seconds: 10 -resources: - gpu_type: gb200 - gpus_per_node: 4 -engine: - type: trtllm - numa_memory_bind: true - served_model_name: nvidia/MiniMax-M3-NVFP4 - publish_events_and_metrics: false -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - env: - HF_HUB_OFFLINE: '1' - TRANSFORMERS_OFFLINE: '1' - HF_HUB_CACHE: /hf_hub_cache - TQDM_DISABLE: '1' - HF_HUB_DISABLE_PROGRESS_BARS: '1' - TLLM_LOG_LEVEL: INFO - TLLM_PROFILE_LOG_RANKS: all - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - PYTHONNOUSERSITE: '1' - TRTLLM_SERVE_ENABLE_MSGSPEC: '1' - TRTLLM_TORCH_COMPILE_CONTEXT_ONLY: '1' - DYN_PUBLISH_KV_EVENTS: '0' - args: - max_seq_len: 1048576 - max_num_tokens: 16384 - max_batch_size: 20 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 3 - - 4 - - 5 - - 6 - - 7 - - 8 - - 9 - - 10 - - 11 - - 12 - - 13 - - 14 - - 15 - - 16 - - 17 - - 18 - - 19 - - 20 - torch_compile_config: - enable_fullgraph: true - enable_inductor: false - enable_piecewise_cuda_graph: true - capture_num_tokens: - - 1 - - 512 - - 1024 - - 2048 - enable_userbuffers: true - max_num_streams: 3 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - sparse_attention_config: - algorithm: minimax_m3 - implementation: msa - indexer_kv_dtype: fp8 - sparse_disable_index_value: true - fuse_qkv_index_projection: true - kv_cache_config: - free_gpu_memory_fraction: 0.94 - enable_block_reuse: true - block_reuse_policy: per_conversation - tokens_per_block: 128 - use_kv_cache_manager_v2: true - dtype: fp8 - event_buffer_max_size: 0 - host_cache_size: 137438953472 - speculative_config: - decoding_type: Eagle3 - max_draft_len: 3 - speculative_model: Inferact/MiniMax-M3-EAGLE3-GQA - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - enable_chunked_prefill: true - enable_autotuner: true - trust_remote_code: true - stream_interval: 20 - print_iter_log: true - num_postprocess_workers: 8 - enable_attention_dp: false - tensor_parallel_size: 4 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_ROUTER_TEMPERATURE: '0' - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TCP_REQUEST_TIMEOUT: '30' - DYN_LOG: warn - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: '0' - SERVED_MODEL_NAME: nvidia/MiniMax-M3-NVFP4 - MAX_MODEL_LEN: '1048576' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - OPENAI_API_KEY: EMPTY - KV_OFFLOADING: dram - KV_OFFLOAD_BACKEND: native - KV_OFFLOAD_BACKEND_METADATA: '{"name":"native"}' - TOTAL_CPU_DRAM_GB: '550' - MODEL: nvidia/MiniMax-M3-NVFP4 - MODEL_PREFIX: minimaxm3 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '20' - DURATION: '3600' diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c25-b25-eagle3.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c25-b25-eagle3.yaml deleted file mode 100644 index 25e2a5df3e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c25-b25-eagle3.yaml +++ /dev/null @@ -1,162 +0,0 @@ -# MiniMax-M3 NVFP4, TensorRT-LLM aggregated TP4 + EAGLE3-GQA (3 draft tokens), one GB200 node, Dynamo frontend. -# Engine config is the max_batch_size=25 member of the TP4 recipe set. -# Host KV cache is set to 128 GiB per rank for the four-rank layout. -# AgentX acceptance is selected from the committed golden curve at submission. -# (= AL - 1 = 1.78 for the committed golden AL 2.78) for throughput runs and leaves eval-only runs on real acceptance. -schema: 2 -name: dynamo-agg-gb200-tp4-c25-b1-eagle3 -model: - path: minimax-m3-nvfp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc23.post1 - precision: fp4 -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - request_plane: tcp -health_check: - max_attempts: 270 - interval_seconds: 10 -resources: - gpu_type: gb200 - gpus_per_node: 4 -engine: - type: trtllm - numa_memory_bind: true - served_model_name: nvidia/MiniMax-M3-NVFP4 - publish_events_and_metrics: false -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - env: - HF_HUB_OFFLINE: '1' - TRANSFORMERS_OFFLINE: '1' - HF_HUB_CACHE: /hf_hub_cache - TQDM_DISABLE: '1' - HF_HUB_DISABLE_PROGRESS_BARS: '1' - TLLM_LOG_LEVEL: INFO - TLLM_PROFILE_LOG_RANKS: all - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - PYTHONNOUSERSITE: '1' - TRTLLM_SERVE_ENABLE_MSGSPEC: '1' - TRTLLM_TORCH_COMPILE_CONTEXT_ONLY: '1' - DYN_PUBLISH_KV_EVENTS: '0' - args: - max_seq_len: 1048576 - max_num_tokens: 16384 - max_batch_size: 25 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 3 - - 4 - - 5 - - 6 - - 7 - - 8 - - 9 - - 10 - - 11 - - 12 - - 13 - - 14 - - 15 - - 17 - - 19 - - 21 - - 23 - - 25 - torch_compile_config: - enable_fullgraph: true - enable_inductor: false - enable_piecewise_cuda_graph: true - capture_num_tokens: - - 1 - - 512 - - 1024 - - 2048 - enable_userbuffers: true - max_num_streams: 3 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - sparse_attention_config: - algorithm: minimax_m3 - implementation: msa - indexer_kv_dtype: fp8 - sparse_disable_index_value: true - fuse_qkv_index_projection: true - kv_cache_config: - free_gpu_memory_fraction: 0.94 - enable_block_reuse: true - block_reuse_policy: per_conversation - tokens_per_block: 128 - use_kv_cache_manager_v2: true - dtype: fp8 - event_buffer_max_size: 0 - host_cache_size: 137438953472 - speculative_config: - decoding_type: Eagle3 - max_draft_len: 3 - speculative_model: Inferact/MiniMax-M3-EAGLE3-GQA - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - enable_chunked_prefill: true - enable_autotuner: true - trust_remote_code: true - stream_interval: 20 - print_iter_log: true - num_postprocess_workers: 8 - enable_attention_dp: false - tensor_parallel_size: 4 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_ROUTER_TEMPERATURE: '0' - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TCP_REQUEST_TIMEOUT: '30' - DYN_LOG: warn - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: '0' - SERVED_MODEL_NAME: nvidia/MiniMax-M3-NVFP4 - MAX_MODEL_LEN: '1048576' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - OPENAI_API_KEY: EMPTY - KV_OFFLOADING: dram - KV_OFFLOAD_BACKEND: native - KV_OFFLOAD_BACKEND_METADATA: '{"name":"native"}' - TOTAL_CPU_DRAM_GB: '550' - MODEL: nvidia/MiniMax-M3-NVFP4 - MODEL_PREFIX: minimaxm3 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '25' - DURATION: '3600' diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c30-b30-eagle3.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c30-b30-eagle3.yaml deleted file mode 100644 index b6315eaa71..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c30-b30-eagle3.yaml +++ /dev/null @@ -1,162 +0,0 @@ -# MiniMax-M3 NVFP4, TensorRT-LLM aggregated TP4 + EAGLE3-GQA (3 draft tokens), one GB200 node, Dynamo frontend. -# Engine config is the max_batch_size=30 member of the TP4 recipe set. -# Host KV cache is set to 128 GiB per rank for the four-rank layout. -# AgentX acceptance is selected from the committed golden curve at submission. -# (= AL - 1 = 1.78 for the committed golden AL 2.78) for throughput runs and leaves eval-only runs on real acceptance. -schema: 2 -name: dynamo-agg-gb200-tp4-c30-b1-eagle3 -model: - path: minimax-m3-nvfp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc23.post1 - precision: fp4 -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - request_plane: tcp -health_check: - max_attempts: 270 - interval_seconds: 10 -resources: - gpu_type: gb200 - gpus_per_node: 4 -engine: - type: trtllm - numa_memory_bind: true - served_model_name: nvidia/MiniMax-M3-NVFP4 - publish_events_and_metrics: false -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - env: - HF_HUB_OFFLINE: '1' - TRANSFORMERS_OFFLINE: '1' - HF_HUB_CACHE: /hf_hub_cache - TQDM_DISABLE: '1' - HF_HUB_DISABLE_PROGRESS_BARS: '1' - TLLM_LOG_LEVEL: INFO - TLLM_PROFILE_LOG_RANKS: all - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - PYTHONNOUSERSITE: '1' - TRTLLM_SERVE_ENABLE_MSGSPEC: '1' - TRTLLM_TORCH_COMPILE_CONTEXT_ONLY: '1' - DYN_PUBLISH_KV_EVENTS: '0' - args: - max_seq_len: 1048576 - max_num_tokens: 16384 - max_batch_size: 30 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 3 - - 4 - - 5 - - 6 - - 7 - - 8 - - 9 - - 10 - - 11 - - 12 - - 14 - - 16 - - 18 - - 20 - - 22 - - 24 - - 27 - - 30 - torch_compile_config: - enable_fullgraph: true - enable_inductor: false - enable_piecewise_cuda_graph: true - capture_num_tokens: - - 1 - - 512 - - 1024 - - 2048 - enable_userbuffers: true - max_num_streams: 3 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - sparse_attention_config: - algorithm: minimax_m3 - implementation: msa - indexer_kv_dtype: fp8 - sparse_disable_index_value: true - fuse_qkv_index_projection: true - kv_cache_config: - free_gpu_memory_fraction: 0.94 - enable_block_reuse: true - block_reuse_policy: per_conversation - tokens_per_block: 128 - use_kv_cache_manager_v2: true - dtype: fp8 - event_buffer_max_size: 0 - host_cache_size: 137438953472 - speculative_config: - decoding_type: Eagle3 - max_draft_len: 3 - speculative_model: Inferact/MiniMax-M3-EAGLE3-GQA - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - enable_chunked_prefill: true - enable_autotuner: true - trust_remote_code: true - stream_interval: 20 - print_iter_log: true - num_postprocess_workers: 8 - enable_attention_dp: false - tensor_parallel_size: 4 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_ROUTER_TEMPERATURE: '0' - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TCP_REQUEST_TIMEOUT: '30' - DYN_LOG: warn - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: '0' - SERVED_MODEL_NAME: nvidia/MiniMax-M3-NVFP4 - MAX_MODEL_LEN: '1048576' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - OPENAI_API_KEY: EMPTY - KV_OFFLOADING: dram - KV_OFFLOAD_BACKEND: native - KV_OFFLOAD_BACKEND_METADATA: '{"name":"native"}' - TOTAL_CPU_DRAM_GB: '550' - MODEL: nvidia/MiniMax-M3-NVFP4 - MODEL_PREFIX: minimaxm3 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '30' - DURATION: '3600' diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c40-b40-eagle3.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c40-b40-eagle3.yaml deleted file mode 100644 index 1d29eebbc1..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c40-b40-eagle3.yaml +++ /dev/null @@ -1,166 +0,0 @@ -# MiniMax-M3 NVFP4, TensorRT-LLM aggregated TP4 + EAGLE3-GQA (3 draft tokens), one GB200 node, Dynamo frontend. -# Engine config is the max_batch_size=40 member of the TP4 recipe set. -# Host KV cache is set to 128 GiB per rank for the four-rank layout. -# AgentX acceptance is selected from the committed golden curve at submission. -# (= AL - 1 = 1.78 for the committed golden AL 2.78) for throughput runs and leaves eval-only runs on real acceptance. -schema: 2 -name: dynamo-agg-gb200-tp4-c40-b1-eagle3 -model: - path: minimax-m3-nvfp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc23.post1 - precision: fp4 -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - request_plane: tcp -health_check: - max_attempts: 270 - interval_seconds: 10 -resources: - gpu_type: gb200 - gpus_per_node: 4 -engine: - type: trtllm - numa_memory_bind: true - served_model_name: nvidia/MiniMax-M3-NVFP4 - publish_events_and_metrics: false -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - env: - HF_HUB_OFFLINE: '1' - TRANSFORMERS_OFFLINE: '1' - HF_HUB_CACHE: /hf_hub_cache - TQDM_DISABLE: '1' - HF_HUB_DISABLE_PROGRESS_BARS: '1' - TLLM_LOG_LEVEL: INFO - TLLM_PROFILE_LOG_RANKS: all - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - PYTHONNOUSERSITE: '1' - TRTLLM_SERVE_ENABLE_MSGSPEC: '1' - TRTLLM_TORCH_COMPILE_CONTEXT_ONLY: '1' - DYN_PUBLISH_KV_EVENTS: '0' - args: - max_seq_len: 1048576 - max_num_tokens: 16384 - max_batch_size: 40 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 3 - - 4 - - 5 - - 6 - - 7 - - 8 - - 9 - - 10 - - 14 - - 16 - - 18 - - 20 - - 22 - - 24 - - 26 - - 28 - - 30 - - 32 - - 34 - - 36 - - 38 - - 40 - torch_compile_config: - enable_fullgraph: true - enable_inductor: false - enable_piecewise_cuda_graph: true - capture_num_tokens: - - 1 - - 512 - - 1024 - - 2048 - enable_userbuffers: true - max_num_streams: 3 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - sparse_attention_config: - algorithm: minimax_m3 - implementation: msa - indexer_kv_dtype: fp8 - sparse_disable_index_value: true - fuse_qkv_index_projection: true - kv_cache_config: - free_gpu_memory_fraction: 0.94 - enable_block_reuse: true - block_reuse_policy: per_conversation - tokens_per_block: 128 - use_kv_cache_manager_v2: true - dtype: fp8 - event_buffer_max_size: 0 - host_cache_size: 137438953472 - speculative_config: - decoding_type: Eagle3 - max_draft_len: 3 - speculative_model: Inferact/MiniMax-M3-EAGLE3-GQA - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - enable_chunked_prefill: true - enable_autotuner: true - trust_remote_code: true - stream_interval: 20 - print_iter_log: true - num_postprocess_workers: 8 - enable_attention_dp: false - tensor_parallel_size: 4 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_ROUTER_TEMPERATURE: '0' - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TCP_REQUEST_TIMEOUT: '30' - DYN_LOG: warn - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: '0' - SERVED_MODEL_NAME: nvidia/MiniMax-M3-NVFP4 - MAX_MODEL_LEN: '1048576' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - OPENAI_API_KEY: EMPTY - KV_OFFLOADING: dram - KV_OFFLOAD_BACKEND: native - KV_OFFLOAD_BACKEND_METADATA: '{"name":"native"}' - TOTAL_CPU_DRAM_GB: '550' - MODEL: nvidia/MiniMax-M3-NVFP4 - MODEL_PREFIX: minimaxm3 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '40' - DURATION: '3600' diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c5-b5-eagle3.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c5-b5-eagle3.yaml deleted file mode 100644 index 9279351de5..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c5-b5-eagle3.yaml +++ /dev/null @@ -1,147 +0,0 @@ -# MiniMax-M3 NVFP4, TensorRT-LLM aggregated TP4 + EAGLE3-GQA (3 draft tokens), one GB200 node, Dynamo frontend. -# Engine config is the max_batch_size=5 member of the TP4 recipe set. -# Host KV cache is set to 128 GiB per rank for the four-rank layout. -# AgentX acceptance is selected from the committed golden curve at submission. -# (= AL - 1 = 1.78 for the committed golden AL 2.78) for throughput runs and leaves eval-only runs on real acceptance. -schema: 2 -name: dynamo-agg-gb200-tp4-c5-b1-eagle3 -model: - path: minimax-m3-nvfp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc23.post1 - precision: fp4 -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - request_plane: tcp -health_check: - max_attempts: 270 - interval_seconds: 10 -resources: - gpu_type: gb200 - gpus_per_node: 4 -engine: - type: trtllm - numa_memory_bind: true - served_model_name: nvidia/MiniMax-M3-NVFP4 - publish_events_and_metrics: false -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - env: - HF_HUB_OFFLINE: '1' - TRANSFORMERS_OFFLINE: '1' - HF_HUB_CACHE: /hf_hub_cache - TQDM_DISABLE: '1' - HF_HUB_DISABLE_PROGRESS_BARS: '1' - TLLM_LOG_LEVEL: INFO - TLLM_PROFILE_LOG_RANKS: all - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - PYTHONNOUSERSITE: '1' - TRTLLM_SERVE_ENABLE_MSGSPEC: '1' - TRTLLM_TORCH_COMPILE_CONTEXT_ONLY: '1' - DYN_PUBLISH_KV_EVENTS: '0' - args: - max_seq_len: 1048576 - max_num_tokens: 16384 - max_batch_size: 5 - cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 3 - - 4 - - 5 - torch_compile_config: - enable_fullgraph: true - enable_inductor: false - enable_piecewise_cuda_graph: true - capture_num_tokens: - - 1 - - 512 - - 1024 - - 2048 - enable_userbuffers: true - max_num_streams: 3 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - sparse_attention_config: - algorithm: minimax_m3 - implementation: msa - indexer_kv_dtype: fp8 - sparse_disable_index_value: true - fuse_qkv_index_projection: true - kv_cache_config: - free_gpu_memory_fraction: 0.94 - enable_block_reuse: true - block_reuse_policy: per_conversation - tokens_per_block: 128 - use_kv_cache_manager_v2: true - dtype: fp8 - event_buffer_max_size: 0 - host_cache_size: 137438953472 - speculative_config: - decoding_type: Eagle3 - max_draft_len: 3 - speculative_model: Inferact/MiniMax-M3-EAGLE3-GQA - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - enable_chunked_prefill: true - enable_autotuner: true - trust_remote_code: true - stream_interval: 20 - print_iter_log: true - num_postprocess_workers: 8 - enable_attention_dp: false - tensor_parallel_size: 4 -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_ROUTER_TEMPERATURE: '0' - DYN_TOKENIZER_CACHE: '1' - DYN_TOKENIZER_CACHE_BYTES: '8000000000' - DYN_TCP_REQUEST_TIMEOUT: '30' - DYN_LOG: warn - DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' - args: - router-mode: kv - no-kv-events: true -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 - AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: '0' - SERVED_MODEL_NAME: nvidia/MiniMax-M3-NVFP4 - MAX_MODEL_LEN: '1048576' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - OPENAI_API_KEY: EMPTY - KV_OFFLOADING: dram - KV_OFFLOAD_BACKEND: native - KV_OFFLOAD_BACKEND_METADATA: '{"name":"native"}' - TOTAL_CPU_DRAM_GB: '550' - MODEL: nvidia/MiniMax-M3-NVFP4 - MODEL_PREFIX: minimaxm3 - FRAMEWORK: dynamo-trt - PRECISION: fp4 - CONC: '5' - DURATION: '3600' diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..8177d5dc2d --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml @@ -0,0 +1,235 @@ +# srt-slurm recipes for minimaxm3/trtllm/gb200-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_. +# +# MiniMax-M3 NVFP4, TensorRT-LLM aggregated TP4 + EAGLE3-GQA (3 draft tokens), one GB200 node, Dynamo frontend. +# Host KV cache is set to 128 GiB per rank for the four-rank layout. +# AgentX acceptance is selected from the committed golden curve at submission. +# (= AL - 1 = 1.78 for the committed golden AL 2.78) for throughput runs and leaves eval-only runs on real acceptance. + +schema: 2 + +base: + model: + path: minimax-m3-nvfp4 + container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc23.post1 + precision: fp4 + dynamo: + install: true + source: + wheel: 1.4.0.dev20260807 + request_plane: tcp + health_check: + max_attempts: 270 + interval_seconds: 10 + resources: + gpu_type: gb200 + gpus_per_node: 4 + engine: + type: trtllm + numa_memory_bind: true + served_model_name: nvidia/MiniMax-M3-NVFP4 + publish_events_and_metrics: false + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + env: + HF_HUB_OFFLINE: '1' + TRANSFORMERS_OFFLINE: '1' + HF_HUB_CACHE: /hf_hub_cache + TQDM_DISABLE: '1' + HF_HUB_DISABLE_PROGRESS_BARS: '1' + TLLM_LOG_LEVEL: INFO + TLLM_PROFILE_LOG_RANKS: all + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + TRTLLM_ENABLE_PDL: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + MIMALLOC_PURGE_DELAY: '0' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + PYTHONNOUSERSITE: '1' + TRTLLM_SERVE_ENABLE_MSGSPEC: '1' + TRTLLM_TORCH_COMPILE_CONTEXT_ONLY: '1' + DYN_PUBLISH_KV_EVENTS: '0' + args: + max_seq_len: 1048576 + max_num_tokens: 16384 + cuda_graph_config: + enable_padding: true + torch_compile_config: + enable_fullgraph: true + enable_inductor: false + enable_piecewise_cuda_graph: true + capture_num_tokens: + - 1 + - 512 + - 1024 + - 2048 + enable_userbuffers: true + max_num_streams: 3 + moe_config: + backend: TRTLLM + use_low_precision_moe_combine: true + sparse_attention_config: + algorithm: minimax_m3 + implementation: msa + indexer_kv_dtype: fp8 + sparse_disable_index_value: true + fuse_qkv_index_projection: true + kv_cache_config: + free_gpu_memory_fraction: 0.94 + enable_block_reuse: true + block_reuse_policy: per_conversation + tokens_per_block: 128 + use_kv_cache_manager_v2: true + dtype: fp8 + event_buffer_max_size: 0 + host_cache_size: 137438953472 + speculative_config: + decoding_type: Eagle3 + max_draft_len: 3 + speculative_model: Inferact/MiniMax-M3-EAGLE3-GQA + scheduler_config: + capacity_scheduler_policy: MAX_UTILIZATION + enable_chunked_prefill: true + enable_autotuner: true + trust_remote_code: true + stream_interval: 20 + print_iter_log: true + num_postprocess_workers: 8 + enable_attention_dp: false + tensor_parallel_size: 4 + frontend: + type: dynamo + enable_multiple_frontends: false + env: + ETCD_LEASE_TTL: '120' + DYN_ROUTER_TEMPERATURE: '0' + DYN_TOKENIZER_CACHE: '1' + DYN_TOKENIZER_CACHE_BYTES: '8000000000' + DYN_TCP_REQUEST_TIMEOUT: '30' + DYN_LOG: warn + DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' + args: + router-mode: kv + no-kv-events: true + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: '0' + SERVED_MODEL_NAME: nvidia/MiniMax-M3-NVFP4 + MAX_MODEL_LEN: '1048576' + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + OPENAI_API_KEY: EMPTY + KV_OFFLOADING: dram + KV_OFFLOAD_BACKEND: native + KV_OFFLOAD_BACKEND_METADATA: '{"name":"native"}' + TOTAL_CPU_DRAM_GB: '550' + MODEL: nvidia/MiniMax-M3-NVFP4 + MODEL_PREFIX: minimaxm3 + FRAMEWORK: dynamo-trt + PRECISION: fp4 + DURATION: '3600' + +# Engine config is the max_batch_size=10 member of the TP4 recipe set. +override_agg_tp4_c10_b10_eagle3: + name: dynamo-agg-gb200-tp4-c10-b1-eagle3 + roles: + agg: + args: + max_batch_size: 10 + cuda_graph_config: + batch_sizes: [1, 2, 3, 4, 5, 6, 7, 8, 9, 10] + benchmark: + env: + CONC: '10' + +# Engine config is the max_batch_size=15 member of the TP4 recipe set. +override_agg_tp4_c15_b15_eagle3: + name: dynamo-agg-gb200-tp4-c15-b1-eagle3 + roles: + agg: + args: + max_batch_size: 15 + cuda_graph_config: + batch_sizes: [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15] + benchmark: + env: + CONC: '15' + +# Engine config is the max_batch_size=20 member of the TP4 recipe set. +override_agg_tp4_c20_b20_eagle3: + name: dynamo-agg-gb200-tp4-c20-b1-eagle3 + roles: + agg: + args: + max_batch_size: 20 + cuda_graph_config: + batch_sizes: [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20] + benchmark: + env: + CONC: '20' + +# Engine config is the max_batch_size=25 member of the TP4 recipe set. +override_agg_tp4_c25_b25_eagle3: + name: dynamo-agg-gb200-tp4-c25-b1-eagle3 + roles: + agg: + args: + max_batch_size: 25 + cuda_graph_config: + batch_sizes: [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 19, 21, 23, 25] + benchmark: + env: + CONC: '25' + +# Engine config is the max_batch_size=30 member of the TP4 recipe set. +override_agg_tp4_c30_b30_eagle3: + name: dynamo-agg-gb200-tp4-c30-b1-eagle3 + roles: + agg: + args: + max_batch_size: 30 + cuda_graph_config: + batch_sizes: [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 14, 16, 18, 20, 22, 24, 27, 30] + benchmark: + env: + CONC: '30' + +# Engine config is the max_batch_size=40 member of the TP4 recipe set. +override_agg_tp4_c40_b40_eagle3: + name: dynamo-agg-gb200-tp4-c40-b1-eagle3 + roles: + agg: + args: + max_batch_size: 40 + cuda_graph_config: + batch_sizes: [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 14, 16, 18, 20, 22, 24, 26, 28, 30, 32, 34, 36, 38, 40] + benchmark: + env: + CONC: '40' + +# Engine config is the max_batch_size=5 member of the TP4 recipe set. +override_agg_tp4_c5_b5_eagle3: + name: dynamo-agg-gb200-tp4-c5-b1-eagle3 + roles: + agg: + args: + max_batch_size: 5 + cuda_graph_config: + batch_sizes: [1, 2, 3, 4, 5] + benchmark: + env: + CONC: '5' diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-nightly-native.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-nightly-native.yaml deleted file mode 100644 index 5b4a692ed4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-nightly-native.yaml +++ /dev/null @@ -1,111 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-agg-gb200-tp4-agentic-nightly-native" - -model: - path: "minimax-m3-nvfp4" - container: "vllm/vllm-openai:nightly-9ea8f3ffc354901b740f0b31988900897b7221d7" - precision: "fp4" - -identity: - model: {repo: "nvidia/MiniMax-M3-NVFP4"} - container: {image: "vllm/vllm-openai:nightly-9ea8f3ffc354901b740f0b31988900897b7221d7"} - frameworks: {dynamo: "1.5.0.dev20260908"} - -dynamo: {install: true, source: {pypi: "1.5.0.dev20260908"}} -environment: {ETCD_LEASE_TTL: "7200"} - -slurm: {time_limit: "12:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - -frontend: - type: dynamo - enable_multiple_frontends: false - env: {DYN_TCP_REQUEST_TIMEOUT: "60"} - args: - trust-remote-code: true - router-mode: "kv" - router-kv-events: true - router-temperature: "0" - router-session-affinity-ttl-secs: 14400 - kv-cache-block-size: 128 - -engine: - type: vllm - connector: -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - - env: - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_LOG_STATS_INTERVAL: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "cuda_copy,cuda_ipc,rc" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 4 - pipeline-parallel-size: 1 - trust-remote-code: true - enable-prefix-caching: true - kv-cache-metrics: true - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE"}' - block-size: 128 - gpu-memory-utilization: 0.9 - max-model-len: 1048576 - language-model-only: true - kv-cache-dtype: "fp8" - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASHINFER"}' - stream-interval: 20 - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - - kv_events: true -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-vllm-simple-nightly-native.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-vllm-simple-nightly-native.yaml deleted file mode 100644 index cd56742486..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-vllm-simple-nightly-native.yaml +++ /dev/null @@ -1,108 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-agg-gb200-tp4-vllm-simple-agentic-nightly-native" - -model: - path: "minimax-m3-nvfp4" - container: "vllm/vllm-openai:nightly-9ea8f3ffc354901b740f0b31988900897b7221d7" - precision: "fp4" - -identity: - model: {repo: "nvidia/MiniMax-M3-NVFP4"} - container: {image: "vllm/vllm-openai:nightly-9ea8f3ffc354901b740f0b31988900897b7221d7"} - frameworks: {dynamo: "1.5.0.dev20260908"} - -dynamo: {install: true, source: {pypi: "1.5.0.dev20260908"}} -environment: {ETCD_LEASE_TTL: "7200"} -slurm: {time_limit: "12:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} -resources: {gpu_type: "gb200", gpus_per_node: 4} -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - -frontend: - type: dynamo - enable_multiple_frontends: false - env: {DYN_TCP_REQUEST_TIMEOUT: "60"} - args: - trust-remote-code: true - router-mode: "kv" - router-kv-events: true - router-temperature: "0" - router-session-affinity-ttl-secs: 14400 - kv-cache-block-size: 128 - -engine: - type: vllm - connector: -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - env: - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_LOG_STATS_INTERVAL: "1" - VLLM_USE_SIMPLE_KV_OFFLOAD: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "cuda_copy,cuda_ipc,rc" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 4 - pipeline-parallel-size: 1 - trust-remote-code: true - enable-prefix-caching: true - kv-cache-metrics: true - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - block-size: 128 - gpu-memory-utilization: 0.9 - max-model-len: 1048576 - language-model-only: true - kv-cache-dtype: "fp8" - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASHINFER"}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":777389080576,"cpu_bytes_to_use_per_rank":194347270144,"lazy_offload":true}}' - stream-interval: 20 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE"}' - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - - kv_events: true -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp8-nightly-native.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp8-nightly-native.yaml deleted file mode 100644 index 590eac5eaf..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp8-nightly-native.yaml +++ /dev/null @@ -1,111 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-agg-gb200-tp8-agentic-nightly-native" - -model: - path: "minimax-m3-nvfp4" - container: "vllm/vllm-openai:nightly-9ea8f3ffc354901b740f0b31988900897b7221d7" - precision: "fp4" - -identity: - model: {repo: "nvidia/MiniMax-M3-NVFP4"} - container: {image: "vllm/vllm-openai:nightly-9ea8f3ffc354901b740f0b31988900897b7221d7"} - frameworks: {dynamo: "1.5.0.dev20260908"} - -dynamo: {install: true, source: {pypi: "1.5.0.dev20260908"}} -environment: {ETCD_LEASE_TTL: "7200"} - -slurm: {time_limit: "12:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - -frontend: - type: dynamo - enable_multiple_frontends: false - env: {DYN_TCP_REQUEST_TIMEOUT: "60"} - args: - trust-remote-code: true - router-mode: "kv" - router-kv-events: true - router-temperature: "0" - router-session-affinity-ttl-secs: 14400 - kv-cache-block-size: 128 - -engine: - type: vllm - connector: -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - - env: - VLLM_ENGINE_READY_TIMEOUT_S: "7200" - VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - VLLM_LOG_STATS_INTERVAL: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - UCX_MEMTYPE_CACHE: "n" - UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" - UCX_TLS: "cuda_copy,cuda_ipc,rc" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 8 - pipeline-parallel-size: 1 - trust-remote-code: true - enable-prefix-caching: true - kv-cache-metrics: true - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - block-size: 128 - gpu-memory-utilization: 0.9 - max-model-len: 1048576 - language-model-only: true - kv-cache-dtype: "fp8" - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASHINFER"}' - stream-interval: 20 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE"}' - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - - kv_events: true -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-tp4-tp4-c24.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-tp4-tp4-c24.yaml deleted file mode 100644 index ea049b7320..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-tp4-tp4-c24.yaml +++ /dev/null @@ -1,173 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb200-1p1d-tp4-tp4-c24-agentic" - -model: - path: "minimax-m3-nvfp4" - container: &container "vllm/vllm-openai@sha256:b9104b7ef3048e42f79fba9ab5da06e5aff8164aca9968692ec2f569aaaf34c6" - precision: "fp4" - -identity: - model: {repo: "nvidia/MiniMax-M3-NVFP4"} - container: {image: *container} - frameworks: {dynamo: "1.5.0.dev20260819"} - -dynamo: {install: true, source: {wheel: "1.5.0.dev20260819"}} -environment: {PYTHONHASHSEED: "0"} -health_check: {max_attempts: 2160, interval_seconds: 10} - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "150GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: true - -frontend: - type: dynamo - enable_multiple_frontends: false - args: - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "minimax_m3" - reasoning-parser: "minimax_m3" - enable-auto-tool-choice: true - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 3600 - env: - DYN_LOG: "info" - DYN_TCP_CONNECT_TIMEOUT: "120" - DYN_TOKENIZER: "fastokens" - DYN_TOKENIZER_CACHE: "1" - -engine: - type: vllm - connector: - dp_launch_mode: per_gpu -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: &worker_environment - HF_HOME: "/hf_hub_cache" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - DYN_TCP_CONNECT_TIMEOUT: "120" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_USE_NCCL_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_MEMTYPE_CACHE: "n" - UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx" - UCX_RCACHE_MAX_UNRELEASED: "1024" - UCX_RNDV_PIPELINE_ERROR_HANDLING: "y" - UCX_TLS: "tcp,cuda_ipc,cuda_copy" - WITH_NVIDIA_PEERMEM: "0" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "1" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 4 - enable-expert-parallel: true - all2all-backend: "flashinfer_nvlink_one_sided" - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - kv-cache-dtype: "fp8" - block-size: 128 - gpu-memory-utilization: 0.9 - max-num-batched-tokens: 16384 - max-cudagraph-capture-size: 512 - stream-interval: 20 - no-enable-flashinfer-autotune: true - enable-cumem-allocator: true - numa-bind: true - numa-bind-nodes: &numa_nodes [0, 0, 1, 1] - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"num_threads":8}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":false,"enable_lookup":false}}]}}' - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - decode: - nodes: 1 - workers: 1 - gpus: 4 - - env: *worker_environment - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 4 - enable-expert-parallel: true - all2all-backend: "flashinfer_nvlink_one_sided" - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - kv-cache-dtype: "fp8" - block-size: 128 - gpu-memory-utilization: 0.9 - max-num-batched-tokens: 16384 - max-cudagraph-capture-size: 512 - stream-interval: 20 - no-enable-flashinfer-autotune: true - enable-cumem-allocator: true - numa-bind: true - numa-bind-nodes: *numa_nodes - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"num_threads":8}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - dyn-default-thinking-mode: "enabled" - -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} - -benchmark: - type: custom - command: "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh" - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AGENTIC_WARMUP_GRACE_PERIOD: "1800" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "3600" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: "1.0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-tp4-tp8-c1.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-tp4-tp8-c1.yaml deleted file mode 100644 index 2a0121782d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-tp4-tp8-c1.yaml +++ /dev/null @@ -1,180 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb200-1p1d-tp4-tp8-c1-agentic" - -model: - path: "minimax-m3-nvfp4" - container: &container "vllm/vllm-openai@sha256:b9104b7ef3048e42f79fba9ab5da06e5aff8164aca9968692ec2f569aaaf34c6" - precision: "fp4" - -identity: - model: {repo: "nvidia/MiniMax-M3-NVFP4"} - container: {image: *container} - frameworks: {dynamo: "1.5.0.dev20260819"} - -dynamo: {install: true, source: {wheel: "1.5.0.dev20260819"}} -environment: {PYTHONHASHSEED: "0"} -health_check: {max_attempts: 2160, interval_seconds: 10} - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "150GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: true - -frontend: - type: dynamo - enable_multiple_frontends: false - args: - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "minimax_m3" - reasoning-parser: "minimax_m3" - enable-auto-tool-choice: true - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 3600 - env: - DYN_LOG: "info" - DYN_TCP_CONNECT_TIMEOUT: "120" - DYN_TOKENIZER: "fastokens" - DYN_TOKENIZER_CACHE: "1" - -engine: - type: vllm - connector: - dp_launch_mode: per_gpu -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: &worker_environment - HF_HOME: "/hf_hub_cache" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - DYN_TCP_CONNECT_TIMEOUT: "120" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_USE_NCCL_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_MEMTYPE_CACHE: "n" - UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx" - UCX_RCACHE_MAX_UNRELEASED: "1024" - UCX_RNDV_PIPELINE_ERROR_HANDLING: "y" - UCX_TLS: "tcp,cuda_ipc,cuda_copy" - WITH_NVIDIA_PEERMEM: "0" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "1" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 4 - enable-expert-parallel: true - all2all-backend: "flashinfer_nvlink_one_sided" - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - kv-cache-dtype: "fp8" - block-size: 128 - gpu-memory-utilization: 0.9 - max-num-batched-tokens: 16384 - max-cudagraph-capture-size: 512 - stream-interval: 20 - no-enable-flashinfer-autotune: true - enable-cumem-allocator: true - numa-bind: true - numa-bind-nodes: &numa_nodes [0, 0, 1, 1] - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"num_threads":8}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":false,"enable_lookup":false}}]}}' - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - decode: - nodes: 2 - workers: 1 - gpus: 8 - - env: - <<: *worker_environment - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "1" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 8 - enable-expert-parallel: true - all2all-backend: "flashinfer_nvlink_one_sided" - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - kv-cache-dtype: "fp8" - block-size: 128 - gpu-memory-utilization: 0.9 - max-num-batched-tokens: 16384 - max-num-seqs: 1 - max-cudagraph-capture-size: 512 - stream-interval: 20 - no-enable-flashinfer-autotune: true - enable-cumem-allocator: true - numa-bind: true - numa-bind-nodes: *numa_nodes - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"num_threads":8}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - dyn-default-thinking-mode: "enabled" - -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} - -benchmark: - type: custom - command: "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh" - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AGENTIC_WARMUP_GRACE_PERIOD: "1800" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "3600" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: "1.0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p2d-tp4-tp4-c8-c16.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p2d-tp4-tp4-c8-c16.yaml deleted file mode 100644 index 513b7a7954..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p2d-tp4-tp4-c8-c16.yaml +++ /dev/null @@ -1,173 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb200-1p2d-tp4-tp4-c8-c16-agentic" - -model: - path: "minimax-m3-nvfp4" - container: &container "vllm/vllm-openai@sha256:b9104b7ef3048e42f79fba9ab5da06e5aff8164aca9968692ec2f569aaaf34c6" - precision: "fp4" - -identity: - model: {repo: "nvidia/MiniMax-M3-NVFP4"} - container: {image: *container} - frameworks: {dynamo: "1.5.0.dev20260819"} - -dynamo: {install: true, source: {wheel: "1.5.0.dev20260819"}} -environment: {PYTHONHASHSEED: "0"} -health_check: {max_attempts: 2160, interval_seconds: 10} - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "150GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: true - -frontend: - type: dynamo - enable_multiple_frontends: false - args: - dyn-chat-processor: "vllm" - trust-remote-code: true - tool-call-parser: "minimax_m3" - reasoning-parser: "minimax_m3" - enable-auto-tool-choice: true - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 3600 - env: - DYN_LOG: "info" - DYN_TCP_CONNECT_TIMEOUT: "120" - DYN_TOKENIZER: "fastokens" - DYN_TOKENIZER_CACHE: "1" - -engine: - type: vllm - connector: - dp_launch_mode: per_gpu -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: &worker_environment - HF_HOME: "/hf_hub_cache" - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - DYN_TCP_CONNECT_TIMEOUT: "120" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - VLLM_USE_NCCL_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_MEMTYPE_CACHE: "n" - UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx" - UCX_RCACHE_MAX_UNRELEASED: "1024" - UCX_RNDV_PIPELINE_ERROR_HANDLING: "y" - UCX_TLS: "tcp,cuda_ipc,cuda_copy" - WITH_NVIDIA_PEERMEM: "0" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "1" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 4 - enable-expert-parallel: true - all2all-backend: "flashinfer_nvlink_one_sided" - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - kv-cache-dtype: "fp8" - block-size: 128 - gpu-memory-utilization: 0.9 - max-num-batched-tokens: 16384 - max-cudagraph-capture-size: 512 - stream-interval: 20 - no-enable-flashinfer-autotune: true - enable-cumem-allocator: true - numa-bind: true - numa-bind-nodes: &numa_nodes [0, 0, 1, 1] - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"num_threads":8}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":false,"enable_lookup":false}}]}}' - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - decode: - nodes: 2 - workers: 2 - gpus: 4 - - env: *worker_environment - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - tensor-parallel-size: 4 - enable-expert-parallel: true - all2all-backend: "flashinfer_nvlink_one_sided" - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - kv-cache-dtype: "fp8" - block-size: 128 - gpu-memory-utilization: 0.9 - max-num-batched-tokens: 16384 - max-cudagraph-capture-size: 512 - stream-interval: 20 - no-enable-flashinfer-autotune: true - enable-cumem-allocator: true - numa-bind: true - numa-bind-nodes: *numa_nodes - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"num_threads":8}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - dyn-default-thinking-mode: "enabled" - -sbatch_directives: {cpus-per-task: "144", mem: "0"} -srun_options: {container-remap-root: ""} - -benchmark: - type: custom - command: "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh" - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AGENTIC_WARMUP_GRACE_PERIOD: "1800" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "3600" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" - AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: "1.0" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..a86c713424 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml @@ -0,0 +1,808 @@ +# srt-slurm recipes for minimaxm3/vllm/gb200-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: minimax-m3-nvfp4 + precision: fp4 + identity: + model: + repo: nvidia/MiniMax-M3-NVFP4 + container: {} + frameworks: {} + dynamo: + install: true + source: {} + environment: {} + health_check: + max_attempts: 2160 + interval_seconds: 10 + resources: + gpu_type: gb200 + gpus_per_node: 4 + frontend: + type: dynamo + enable_multiple_frontends: false + env: {} + args: + trust-remote-code: true + engine: + type: vllm + connector: null + roles: {} + sbatch_directives: + cpus-per-task: '144' + mem: '0' + srun_options: + container-remap-root: '' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + AIPERF_EXTRA_INPUTS: thinking:true + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + +override_agg_tp4_nightly_native: + name: minimax-m3-vllm-agg-gb200-tp4-agentic-nightly-native + model: + container: vllm/vllm-openai:nightly-9ea8f3ffc354901b740f0b31988900897b7221d7 + identity: + container: + image: vllm/vllm-openai:nightly-9ea8f3ffc354901b740f0b31988900897b7221d7 + frameworks: + dynamo: 1.5.0.dev20260908 + dynamo: + source: + pypi: 1.5.0.dev20260908 + environment: + ETCD_LEASE_TTL: '7200' + slurm: + time_limit: '12:00:00' + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + env: + DYN_TCP_REQUEST_TIMEOUT: '60' + args: + router-mode: kv + router-kv-events: true + router-temperature: '0' + router-session-affinity-ttl-secs: 14400 + kv-cache-block-size: 128 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '7200' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_FLOAT32_MATMUL_PRECISION: high + VLLM_FLASHINFER_ALLREDUCE_BACKEND: trtllm + VLLM_LOG_STATS_INTERVAL: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: cuda_copy,cuda_ipc,rc + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + args: + served-model-name: nvidia/MiniMax-M3-NVFP4 + tensor-parallel-size: 4 + pipeline-parallel-size: 1 + trust-remote-code: true + enable-prefix-caching: true + kv-cache-metrics: true + attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE"}' + block-size: 128 + gpu-memory-utilization: 0.9 + max-model-len: 1048576 + language-model-only: true + kv-cache-dtype: fp8 + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASHINFER"}' + stream-interval: 20 + max-cudagraph-capture-size: 512 + max-num-batched-tokens: 16384 + no-enable-flashinfer-autotune: true + reasoning-parser: minimax_m3 + dyn-tool-call-parser: minimax_m3 + dyn-reasoning-parser: minimax_m3 + kv_events: true + benchmark: + env: + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + +override_agg_tp4_vllm_simple_nightly_native: + name: minimax-m3-vllm-agg-gb200-tp4-vllm-simple-agentic-nightly-native + model: + container: vllm/vllm-openai:nightly-9ea8f3ffc354901b740f0b31988900897b7221d7 + identity: + container: + image: vllm/vllm-openai:nightly-9ea8f3ffc354901b740f0b31988900897b7221d7 + frameworks: + dynamo: 1.5.0.dev20260908 + dynamo: + source: + pypi: 1.5.0.dev20260908 + environment: + ETCD_LEASE_TTL: '7200' + slurm: + time_limit: '12:00:00' + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + env: + DYN_TCP_REQUEST_TIMEOUT: '60' + args: + router-mode: kv + router-kv-events: true + router-temperature: '0' + router-session-affinity-ttl-secs: 14400 + kv-cache-block-size: 128 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '7200' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_FLOAT32_MATMUL_PRECISION: high + VLLM_FLASHINFER_ALLREDUCE_BACKEND: trtllm + VLLM_LOG_STATS_INTERVAL: '1' + VLLM_USE_SIMPLE_KV_OFFLOAD: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: cuda_copy,cuda_ipc,rc + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + args: + served-model-name: nvidia/MiniMax-M3-NVFP4 + tensor-parallel-size: 4 + pipeline-parallel-size: 1 + trust-remote-code: true + enable-prefix-caching: true + kv-cache-metrics: true + attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' + block-size: 128 + gpu-memory-utilization: 0.9 + max-model-len: 1048576 + language-model-only: true + kv-cache-dtype: fp8 + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASHINFER"}' + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":777389080576,"cpu_bytes_to_use_per_rank":194347270144,"lazy_offload":true}}' + stream-interval: 20 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE"}' + max-cudagraph-capture-size: 512 + max-num-batched-tokens: 16384 + no-enable-flashinfer-autotune: true + reasoning-parser: minimax_m3 + dyn-tool-call-parser: minimax_m3 + dyn-reasoning-parser: minimax_m3 + kv_events: true + benchmark: + env: + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + +override_agg_tp8_nightly_native: + name: minimax-m3-vllm-agg-gb200-tp8-agentic-nightly-native + model: + container: vllm/vllm-openai:nightly-9ea8f3ffc354901b740f0b31988900897b7221d7 + identity: + container: + image: vllm/vllm-openai:nightly-9ea8f3ffc354901b740f0b31988900897b7221d7 + frameworks: + dynamo: 1.5.0.dev20260908 + dynamo: + source: + pypi: 1.5.0.dev20260908 + environment: + ETCD_LEASE_TTL: '7200' + slurm: + time_limit: '12:00:00' + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + frontend: + env: + DYN_TCP_REQUEST_TIMEOUT: '60' + args: + router-mode: kv + router-kv-events: true + router-temperature: '0' + router-session-affinity-ttl-secs: 14400 + kv-cache-block-size: 128 + roles: + agg: + nodes: 2 + workers: 1 + gpus: 8 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '7200' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' + VLLM_FLOAT32_MATMUL_PRECISION: high + VLLM_FLASHINFER_ALLREDUCE_BACKEND: mnnvl + VLLM_LOG_STATS_INTERVAL: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + UCX_MEMTYPE_CACHE: n + UCX_NET_DEVICES: mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1 + UCX_TLS: cuda_copy,cuda_ipc,rc + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + args: + served-model-name: nvidia/MiniMax-M3-NVFP4 + tensor-parallel-size: 8 + pipeline-parallel-size: 1 + trust-remote-code: true + enable-prefix-caching: true + kv-cache-metrics: true + attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' + block-size: 128 + gpu-memory-utilization: 0.9 + max-model-len: 1048576 + language-model-only: true + kv-cache-dtype: fp8 + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASHINFER"}' + stream-interval: 20 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE"}' + max-cudagraph-capture-size: 512 + max-num-batched-tokens: 16384 + no-enable-flashinfer-autotune: true + reasoning-parser: minimax_m3 + dyn-tool-call-parser: minimax_m3 + dyn-reasoning-parser: minimax_m3 + kv_events: true + benchmark: + env: + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400' + +override_disagg_1p1d_tp4_tp4_c24: + name: minimax-m3-vllm-disagg-gb200-1p1d-tp4-tp4-c24-agentic + model: + container: vllm/vllm-openai@sha256:b9104b7ef3048e42f79fba9ab5da06e5aff8164aca9968692ec2f569aaaf34c6 + identity: + container: + image: vllm/vllm-openai@sha256:b9104b7ef3048e42f79fba9ab5da06e5aff8164aca9968692ec2f569aaaf34c6 + frameworks: + dynamo: 1.5.0.dev20260819 + dynamo: + source: + wheel: 1.5.0.dev20260819 + environment: + PYTHONHASHSEED: '0' + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 150GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: true + frontend: + env: + DYN_LOG: info + DYN_TCP_CONNECT_TIMEOUT: '120' + DYN_TOKENIZER: fastokens + DYN_TOKENIZER_CACHE: '1' + args: + router-mode: least-loaded + router-session-affinity-ttl-secs: 3600 + dyn-chat-processor: vllm + tool-call-parser: minimax_m3 + reasoning-parser: minimax_m3 + enable-auto-tool-choice: true + engine: + dp_launch_mode: per_gpu + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + HF_HOME: /hf_hub_cache + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + DYN_TCP_CONNECT_TIMEOUT: '120' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_FLOAT32_MATMUL_PRECISION: high + VLLM_FLASHINFER_ALLREDUCE_BACKEND: trtllm + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_USE_NCCL_SYMM_MEM: '0' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + UCX_CUDA_IPC_ENABLE_MNNVL: y + UCX_MEMTYPE_CACHE: n + UCX_MODULE_DIR: /usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_RNDV_PIPELINE_ERROR_HANDLING: y + UCX_TLS: tcp,cuda_ipc,cuda_copy + WITH_NVIDIA_PEERMEM: '0' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '1' + args: + served-model-name: nvidia/MiniMax-M3-NVFP4 + tensor-parallel-size: 4 + enable-expert-parallel: true + all2all-backend: flashinfer_nvlink_one_sided + trust-remote-code: true + enable-prefix-caching: true + language-model-only: true + kv-cache-dtype: fp8 + block-size: 128 + gpu-memory-utilization: 0.9 + max-num-batched-tokens: 16384 + max-cudagraph-capture-size: 512 + stream-interval: 20 + no-enable-flashinfer-autotune: true + enable-cumem-allocator: true + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"num_threads":8}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":false,"enable_lookup":false}}]}}' + reasoning-parser: minimax_m3 + dyn-tool-call-parser: minimax_m3 + dyn-reasoning-parser: minimax_m3 + decode: + nodes: 1 + workers: 1 + gpus: 4 + env: + HF_HOME: /hf_hub_cache + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + DYN_TCP_CONNECT_TIMEOUT: '120' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_FLOAT32_MATMUL_PRECISION: high + VLLM_FLASHINFER_ALLREDUCE_BACKEND: trtllm + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_USE_NCCL_SYMM_MEM: '0' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + UCX_CUDA_IPC_ENABLE_MNNVL: y + UCX_MEMTYPE_CACHE: n + UCX_MODULE_DIR: /usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_RNDV_PIPELINE_ERROR_HANDLING: y + UCX_TLS: tcp,cuda_ipc,cuda_copy + WITH_NVIDIA_PEERMEM: '0' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '1' + args: + served-model-name: nvidia/MiniMax-M3-NVFP4 + tensor-parallel-size: 4 + enable-expert-parallel: true + all2all-backend: flashinfer_nvlink_one_sided + trust-remote-code: true + enable-prefix-caching: true + language-model-only: true + kv-cache-dtype: fp8 + block-size: 128 + gpu-memory-utilization: 0.9 + max-num-batched-tokens: 16384 + max-cudagraph-capture-size: 512 + stream-interval: 20 + no-enable-flashinfer-autotune: true + enable-cumem-allocator: true + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"num_threads":8}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' + reasoning-parser: minimax_m3 + dyn-tool-call-parser: minimax_m3 + dyn-reasoning-parser: minimax_m3 + dyn-default-thinking-mode: enabled + benchmark: + env: + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '3600' + AGENTIC_WARMUP_GRACE_PERIOD: '1800' + AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: '1.0' + +override_disagg_1p1d_tp4_tp8_c1: + name: minimax-m3-vllm-disagg-gb200-1p1d-tp4-tp8-c1-agentic + model: + container: vllm/vllm-openai@sha256:b9104b7ef3048e42f79fba9ab5da06e5aff8164aca9968692ec2f569aaaf34c6 + identity: + container: + image: vllm/vllm-openai@sha256:b9104b7ef3048e42f79fba9ab5da06e5aff8164aca9968692ec2f569aaaf34c6 + frameworks: + dynamo: 1.5.0.dev20260819 + dynamo: + source: + wheel: 1.5.0.dev20260819 + environment: + PYTHONHASHSEED: '0' + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 150GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: true + frontend: + env: + DYN_LOG: info + DYN_TCP_CONNECT_TIMEOUT: '120' + DYN_TOKENIZER: fastokens + DYN_TOKENIZER_CACHE: '1' + args: + router-mode: least-loaded + router-session-affinity-ttl-secs: 3600 + dyn-chat-processor: vllm + tool-call-parser: minimax_m3 + reasoning-parser: minimax_m3 + enable-auto-tool-choice: true + engine: + dp_launch_mode: per_gpu + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + HF_HOME: /hf_hub_cache + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + DYN_TCP_CONNECT_TIMEOUT: '120' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_FLOAT32_MATMUL_PRECISION: high + VLLM_FLASHINFER_ALLREDUCE_BACKEND: trtllm + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_USE_NCCL_SYMM_MEM: '0' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + UCX_CUDA_IPC_ENABLE_MNNVL: y + UCX_MEMTYPE_CACHE: n + UCX_MODULE_DIR: /usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_RNDV_PIPELINE_ERROR_HANDLING: y + UCX_TLS: tcp,cuda_ipc,cuda_copy + WITH_NVIDIA_PEERMEM: '0' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '1' + args: + served-model-name: nvidia/MiniMax-M3-NVFP4 + tensor-parallel-size: 4 + enable-expert-parallel: true + all2all-backend: flashinfer_nvlink_one_sided + trust-remote-code: true + enable-prefix-caching: true + language-model-only: true + kv-cache-dtype: fp8 + block-size: 128 + gpu-memory-utilization: 0.9 + max-num-batched-tokens: 16384 + max-cudagraph-capture-size: 512 + stream-interval: 20 + no-enable-flashinfer-autotune: true + enable-cumem-allocator: true + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"num_threads":8}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":false,"enable_lookup":false}}]}}' + reasoning-parser: minimax_m3 + dyn-tool-call-parser: minimax_m3 + dyn-reasoning-parser: minimax_m3 + decode: + nodes: 2 + workers: 1 + gpus: 8 + env: + HF_HOME: /hf_hub_cache + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + DYN_TCP_CONNECT_TIMEOUT: '120' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_FLOAT32_MATMUL_PRECISION: high + VLLM_FLASHINFER_ALLREDUCE_BACKEND: mnnvl + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_USE_NCCL_SYMM_MEM: '0' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + UCX_CUDA_IPC_ENABLE_MNNVL: y + UCX_MEMTYPE_CACHE: n + UCX_MODULE_DIR: /usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_RNDV_PIPELINE_ERROR_HANDLING: y + UCX_TLS: tcp,cuda_ipc,cuda_copy + WITH_NVIDIA_PEERMEM: '0' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '1' + args: + served-model-name: nvidia/MiniMax-M3-NVFP4 + tensor-parallel-size: 8 + enable-expert-parallel: true + all2all-backend: flashinfer_nvlink_one_sided + trust-remote-code: true + enable-prefix-caching: true + language-model-only: true + kv-cache-dtype: fp8 + block-size: 128 + gpu-memory-utilization: 0.9 + max-num-batched-tokens: 16384 + max-num-seqs: 1 + max-cudagraph-capture-size: 512 + stream-interval: 20 + no-enable-flashinfer-autotune: true + enable-cumem-allocator: true + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"num_threads":8}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' + reasoning-parser: minimax_m3 + dyn-tool-call-parser: minimax_m3 + dyn-reasoning-parser: minimax_m3 + dyn-default-thinking-mode: enabled + benchmark: + env: + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '3600' + AGENTIC_WARMUP_GRACE_PERIOD: '1800' + AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: '1.0' + +override_disagg_1p2d_tp4_tp4_c8_c16: + name: minimax-m3-vllm-disagg-gb200-1p2d-tp4-tp4-c8-c16-agentic + model: + container: vllm/vllm-openai@sha256:b9104b7ef3048e42f79fba9ab5da06e5aff8164aca9968692ec2f569aaaf34c6 + identity: + container: + image: vllm/vllm-openai@sha256:b9104b7ef3048e42f79fba9ab5da06e5aff8164aca9968692ec2f569aaaf34c6 + frameworks: + dynamo: 1.5.0.dev20260819 + dynamo: + source: + wheel: 1.5.0.dev20260819 + environment: + PYTHONHASHSEED: '0' + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 150GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: true + frontend: + env: + DYN_LOG: info + DYN_TCP_CONNECT_TIMEOUT: '120' + DYN_TOKENIZER: fastokens + DYN_TOKENIZER_CACHE: '1' + args: + router-mode: least-loaded + router-session-affinity-ttl-secs: 3600 + dyn-chat-processor: vllm + tool-call-parser: minimax_m3 + reasoning-parser: minimax_m3 + enable-auto-tool-choice: true + engine: + dp_launch_mode: per_gpu + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + HF_HOME: /hf_hub_cache + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + DYN_TCP_CONNECT_TIMEOUT: '120' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_FLOAT32_MATMUL_PRECISION: high + VLLM_FLASHINFER_ALLREDUCE_BACKEND: trtllm + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_USE_NCCL_SYMM_MEM: '0' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + UCX_CUDA_IPC_ENABLE_MNNVL: y + UCX_MEMTYPE_CACHE: n + UCX_MODULE_DIR: /usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_RNDV_PIPELINE_ERROR_HANDLING: y + UCX_TLS: tcp,cuda_ipc,cuda_copy + WITH_NVIDIA_PEERMEM: '0' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '1' + args: + served-model-name: nvidia/MiniMax-M3-NVFP4 + tensor-parallel-size: 4 + enable-expert-parallel: true + all2all-backend: flashinfer_nvlink_one_sided + trust-remote-code: true + enable-prefix-caching: true + language-model-only: true + kv-cache-dtype: fp8 + block-size: 128 + gpu-memory-utilization: 0.9 + max-num-batched-tokens: 16384 + max-cudagraph-capture-size: 512 + stream-interval: 20 + no-enable-flashinfer-autotune: true + enable-cumem-allocator: true + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"num_threads":8}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":false,"enable_lookup":false}}]}}' + reasoning-parser: minimax_m3 + dyn-tool-call-parser: minimax_m3 + dyn-reasoning-parser: minimax_m3 + decode: + nodes: 2 + workers: 2 + gpus: 4 + env: + HF_HOME: /hf_hub_cache + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + DYN_TCP_CONNECT_TIMEOUT: '120' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_FLOAT32_MATMUL_PRECISION: high + VLLM_FLASHINFER_ALLREDUCE_BACKEND: trtllm + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + VLLM_USE_NCCL_SYMM_MEM: '0' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + UCX_CUDA_IPC_ENABLE_MNNVL: y + UCX_MEMTYPE_CACHE: n + UCX_MODULE_DIR: /usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx + UCX_RCACHE_MAX_UNRELEASED: '1024' + UCX_RNDV_PIPELINE_ERROR_HANDLING: y + UCX_TLS: tcp,cuda_ipc,cuda_copy + WITH_NVIDIA_PEERMEM: '0' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '1' + args: + served-model-name: nvidia/MiniMax-M3-NVFP4 + tensor-parallel-size: 4 + enable-expert-parallel: true + all2all-backend: flashinfer_nvlink_one_sided + trust-remote-code: true + enable-prefix-caching: true + language-model-only: true + kv-cache-dtype: fp8 + block-size: 128 + gpu-memory-utilization: 0.9 + max-num-batched-tokens: 16384 + max-cudagraph-capture-size: 512 + stream-interval: 20 + no-enable-flashinfer-autotune: true + enable-cumem-allocator: true + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"num_threads":8}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' + reasoning-parser: minimax_m3 + dyn-tool-call-parser: minimax_m3 + dyn-reasoning-parser: minimax_m3 + dyn-default-thinking-mode: enabled + benchmark: + env: + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '3600' + AGENTIC_WARMUP_GRACE_PERIOD: '1800' + AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: '1.0' diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tep4-tp4-c1-eval.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tep4-tp4-c1-eval.yaml deleted file mode 100644 index 44e9395766..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tep4-tp4-c1-eval.yaml +++ /dev/null @@ -1,179 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb300-1p1d-tep4-tp4-c1-fp4-eval-agentic" - -model: - path: "nvidia/MiniMax-M3-NVFP4" - container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - precision: "fp4" - -identity: - model: - repo: "nvidia/MiniMax-M3-NVFP4" - container: - image: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - frameworks: - dynamo: "1.4.0.dev20260730" - -dynamo: - install: true - - source: - pypi: "1.4.0.dev20260730" -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: true -environment: - PYTHONHASHSEED: "0" - -frontend: - type: dynamo - enable_multiple_frontends: false - nginx_session_affinity: true - nginx_session_affinity_header: "X-Dynamo-Session-ID" - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 1800 - env: - DYN_TOKENIZER: "fastokens" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - DYN_TCP_CONNECT_TIMEOUT: "120" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: &worker_environment - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - DYN_TCP_CONNECT_TIMEOUT: "120" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_USE_NCCL_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - UCX_MEMTYPE_CACHE: "n" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx" - UCX_RNDV_PIPELINE_ERROR_HANDLING: "y" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 4 - enable-expert-parallel: true - all2all-backend: "flashinfer_nvlink_one_sided" - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":true}}]}}' - enable-cumem-allocator: true - decode: - nodes: 1 - workers: 1 - gpus: 4 - - env: - <<: *worker_environment - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - dyn-default-thinking-mode: "enabled" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - max-num-seqs: 1024 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 4 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' - enable-cumem-allocator: true -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh" - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: "1.0" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tep4-tp4-c1.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tep4-tp4-c1.yaml deleted file mode 100644 index d121a66c8b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tep4-tp4-c1.yaml +++ /dev/null @@ -1,179 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb300-1p1d-tep4-tp4-c1-fp4-agentic" - -model: - path: "nvidia/MiniMax-M3-NVFP4" - container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - precision: "fp4" - -identity: - model: - repo: "nvidia/MiniMax-M3-NVFP4" - container: - image: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - frameworks: - dynamo: "1.4.0.dev20260730" - -dynamo: - install: true - - source: - pypi: "1.4.0.dev20260730" -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: true -environment: - PYTHONHASHSEED: "0" - -frontend: - type: dynamo - enable_multiple_frontends: false - nginx_session_affinity: true - nginx_session_affinity_header: "X-Dynamo-Session-ID" - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 1800 - env: - DYN_TOKENIZER: "fastokens" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - DYN_TCP_CONNECT_TIMEOUT: "120" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: &worker_environment - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - DYN_TCP_CONNECT_TIMEOUT: "120" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_USE_NCCL_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - UCX_MEMTYPE_CACHE: "n" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx" - UCX_RNDV_PIPELINE_ERROR_HANDLING: "y" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 4 - enable-expert-parallel: true - all2all-backend: "flashinfer_nvlink_one_sided" - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":true}}]}}' - enable-cumem-allocator: true - decode: - nodes: 1 - workers: 1 - gpus: 4 - - env: - <<: *worker_environment - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - dyn-default-thinking-mode: "enabled" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - max-num-seqs: 1024 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 4 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' - enable-cumem-allocator: true -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh" - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: "1.0" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tp2-tp4-c20-c24-eval.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tp2-tp4-c20-c24-eval.yaml deleted file mode 100644 index 88bc24e782..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tp2-tp4-c20-c24-eval.yaml +++ /dev/null @@ -1,179 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb300-1p1d-tp2-tp4-c20-c24-fp4-eval-agentic" - -model: - path: "nvidia/MiniMax-M3-NVFP4" - container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - precision: "fp4" - -identity: - model: - repo: "nvidia/MiniMax-M3-NVFP4" - container: - image: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - frameworks: - dynamo: "1.4.0.dev20260730" - -dynamo: - install: true - - source: - pypi: "1.4.0.dev20260730" -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: true -environment: - PYTHONHASHSEED: "0" - -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 1 - nginx_session_affinity: true - nginx_session_affinity_header: "X-Dynamo-Session-ID" - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 1800 - env: - DYN_TOKENIZER: "fastokens" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - DYN_TCP_CONNECT_TIMEOUT: "120" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - env: &worker_environment - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - DYN_TCP_CONNECT_TIMEOUT: "120" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_USE_NCCL_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - UCX_MEMTYPE_CACHE: "n" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx" - UCX_RNDV_PIPELINE_ERROR_HANDLING: "y" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 2 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":true}}]}}' - enable-cumem-allocator: true - decode: - nodes: 1 - workers: 1 - gpus: 4 - - env: - <<: *worker_environment - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - dyn-default-thinking-mode: "enabled" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - max-num-seqs: 1024 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 4 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' - enable-cumem-allocator: true -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh" - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: "1.0" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tp2-tp4-c20-c24.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tp2-tp4-c20-c24.yaml deleted file mode 100644 index 3677f72f71..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tp2-tp4-c20-c24.yaml +++ /dev/null @@ -1,179 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb300-1p1d-tp2-tp4-c20-c24-fp4-agentic" - -model: - path: "nvidia/MiniMax-M3-NVFP4" - container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - precision: "fp4" - -identity: - model: - repo: "nvidia/MiniMax-M3-NVFP4" - container: - image: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - frameworks: - dynamo: "1.4.0.dev20260730" - -dynamo: - install: true - - source: - pypi: "1.4.0.dev20260730" -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: true -environment: - PYTHONHASHSEED: "0" - -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 1 - nginx_session_affinity: true - nginx_session_affinity_header: "X-Dynamo-Session-ID" - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 1800 - env: - DYN_TOKENIZER: "fastokens" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - DYN_TCP_CONNECT_TIMEOUT: "120" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - env: &worker_environment - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - DYN_TCP_CONNECT_TIMEOUT: "120" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_USE_NCCL_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - UCX_MEMTYPE_CACHE: "n" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx" - UCX_RNDV_PIPELINE_ERROR_HANDLING: "y" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 2 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":true}}]}}' - enable-cumem-allocator: true - decode: - nodes: 1 - workers: 1 - gpus: 4 - - env: - <<: *worker_environment - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - dyn-default-thinking-mode: "enabled" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - max-num-seqs: 1024 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 4 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' - enable-cumem-allocator: true -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh" - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: "1.0" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-dep4-tp4-c24-eval.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-dep4-tp4-c24-eval.yaml deleted file mode 100644 index 1e6fadc381..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-dep4-tp4-c24-eval.yaml +++ /dev/null @@ -1,182 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb300-1p3d-dep4-tp4-c24-fp4-eval-agentic" - -model: - path: "nvidia/MiniMax-M3-NVFP4" - container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - precision: "fp4" - -identity: - model: - repo: "nvidia/MiniMax-M3-NVFP4" - container: - image: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - frameworks: - dynamo: "1.4.0.dev20260730" - -dynamo: - install: true - - source: - pypi: "1.4.0.dev20260730" -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: true -environment: - PYTHONHASHSEED: "0" - -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 1 - nginx_session_affinity: true - nginx_session_affinity_header: "X-Dynamo-Session-ID" - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 1800 - env: - DYN_TOKENIZER: "fastokens" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - DYN_TCP_CONNECT_TIMEOUT: "120" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: &worker_environment - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - DYN_TCP_CONNECT_TIMEOUT: "120" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_USE_NCCL_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - UCX_MEMTYPE_CACHE: "n" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx" - UCX_RNDV_PIPELINE_ERROR_HANDLING: "y" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 1 - data-parallel-size: 4 - data-parallel-rpc-port: 13345 - enable-expert-parallel: true - all2all-backend: "flashinfer_nvlink_one_sided" - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":true}}]}}' - enable-cumem-allocator: true - decode: - nodes: 3 - workers: 3 - gpus: 4 - - env: - <<: *worker_environment - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - dyn-default-thinking-mode: "enabled" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - max-num-seqs: 1024 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 4 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' - enable-cumem-allocator: true -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh" - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: "1.0" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-dep4-tp4-c24.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-dep4-tp4-c24.yaml deleted file mode 100644 index 7efc02a2ab..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-dep4-tp4-c24.yaml +++ /dev/null @@ -1,182 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb300-1p3d-dep4-tp4-c24-fp4-agentic" - -model: - path: "nvidia/MiniMax-M3-NVFP4" - container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - precision: "fp4" - -identity: - model: - repo: "nvidia/MiniMax-M3-NVFP4" - container: - image: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - frameworks: - dynamo: "1.4.0.dev20260730" - -dynamo: - install: true - - source: - pypi: "1.4.0.dev20260730" -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: false -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: true -environment: - PYTHONHASHSEED: "0" - -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 1 - nginx_session_affinity: true - nginx_session_affinity_header: "X-Dynamo-Session-ID" - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 1800 - env: - DYN_TOKENIZER: "fastokens" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - DYN_TCP_CONNECT_TIMEOUT: "120" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: &worker_environment - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - DYN_TCP_CONNECT_TIMEOUT: "120" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_USE_NCCL_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - UCX_MEMTYPE_CACHE: "n" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx" - UCX_RNDV_PIPELINE_ERROR_HANDLING: "y" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 1 - data-parallel-size: 4 - data-parallel-rpc-port: 13345 - enable-expert-parallel: true - all2all-backend: "flashinfer_nvlink_one_sided" - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":true}}]}}' - enable-cumem-allocator: true - decode: - nodes: 3 - workers: 3 - gpus: 4 - - env: - <<: *worker_environment - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - dyn-default-thinking-mode: "enabled" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - max-num-seqs: 1024 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 4 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' - enable-cumem-allocator: true -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh" - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: "1.0" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-tp2-tp2-c48-eval.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-tp2-tp2-c48-eval.yaml deleted file mode 100644 index d0e7aa0acf..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-tp2-tp2-c48-eval.yaml +++ /dev/null @@ -1,179 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb300-1p3d-tp2-tp2-c48-fp4-eval-agentic" - -model: - path: "nvidia/MiniMax-M3-NVFP4" - container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - precision: "fp4" - -identity: - model: - repo: "nvidia/MiniMax-M3-NVFP4" - container: - image: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - frameworks: - dynamo: "1.4.0.dev20260730" - -dynamo: - install: true - - source: - pypi: "1.4.0.dev20260730" -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: true -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: true -environment: - PYTHONHASHSEED: "0" - -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 4 - nginx_session_affinity: true - nginx_session_affinity_header: "X-Dynamo-Session-ID" - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 1800 - env: - DYN_TOKENIZER: "fastokens" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - DYN_TCP_CONNECT_TIMEOUT: "120" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - env: &worker_environment - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - DYN_TCP_CONNECT_TIMEOUT: "120" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_USE_NCCL_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - UCX_MEMTYPE_CACHE: "n" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx" - UCX_RNDV_PIPELINE_ERROR_HANDLING: "y" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 2 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":true}}]}}' - enable-cumem-allocator: true - decode: - nodes: 3 - workers: 3 - gpus: 2 - - env: - <<: *worker_environment - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - dyn-default-thinking-mode: "enabled" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - max-num-seqs: 1024 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 2 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' - enable-cumem-allocator: true -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh" - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: "1.0" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-tp2-tp2-c48.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-tp2-tp2-c48.yaml deleted file mode 100644 index 142bad0347..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-tp2-tp2-c48.yaml +++ /dev/null @@ -1,179 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb300-1p3d-tp2-tp2-c48-fp4-agentic" - -model: - path: "nvidia/MiniMax-M3-NVFP4" - container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - precision: "fp4" - -identity: - model: - repo: "nvidia/MiniMax-M3-NVFP4" - container: - image: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - frameworks: - dynamo: "1.4.0.dev20260730" - -dynamo: - install: true - - source: - pypi: "1.4.0.dev20260730" -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: true -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: true -environment: - PYTHONHASHSEED: "0" - -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 4 - nginx_session_affinity: true - nginx_session_affinity_header: "X-Dynamo-Session-ID" - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 1800 - env: - DYN_TOKENIZER: "fastokens" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - DYN_TCP_CONNECT_TIMEOUT: "120" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - env: &worker_environment - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - DYN_TCP_CONNECT_TIMEOUT: "120" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_USE_NCCL_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - UCX_MEMTYPE_CACHE: "n" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx" - UCX_RNDV_PIPELINE_ERROR_HANDLING: "y" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 2 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":true}}]}}' - enable-cumem-allocator: true - decode: - nodes: 3 - workers: 3 - gpus: 2 - - env: - <<: *worker_environment - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - dyn-default-thinking-mode: "enabled" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - max-num-seqs: 1024 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 2 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' - enable-cumem-allocator: true -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh" - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: "1.0" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-2p5d-tp2-tp2-c120-eval.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-2p5d-tp2-tp2-c120-eval.yaml deleted file mode 100644 index 2ca720c02e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-2p5d-tp2-tp2-c120-eval.yaml +++ /dev/null @@ -1,179 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb300-2p5d-tp2-tp2-c120-fp4-eval-agentic" - -model: - path: "nvidia/MiniMax-M3-NVFP4" - container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - precision: "fp4" - -identity: - model: - repo: "nvidia/MiniMax-M3-NVFP4" - container: - image: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - frameworks: - dynamo: "1.4.0.dev20260730" - -dynamo: - install: true - - source: - pypi: "1.4.0.dev20260730" -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: true -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: true -environment: - PYTHONHASHSEED: "0" - -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 4 - nginx_session_affinity: true - nginx_session_affinity_header: "X-Dynamo-Session-ID" - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 1800 - env: - DYN_TOKENIZER: "fastokens" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - DYN_TCP_CONNECT_TIMEOUT: "120" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 2 - workers: 2 - gpus: 2 - env: &worker_environment - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - DYN_TCP_CONNECT_TIMEOUT: "120" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_USE_NCCL_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - UCX_MEMTYPE_CACHE: "n" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx" - UCX_RNDV_PIPELINE_ERROR_HANDLING: "y" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 2 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":true}}]}}' - enable-cumem-allocator: true - decode: - nodes: 5 - workers: 5 - gpus: 2 - - env: - <<: *worker_environment - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - dyn-default-thinking-mode: "enabled" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - max-num-seqs: 1024 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 2 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' - enable-cumem-allocator: true -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh" - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: "1.0" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-2p5d-tp2-tp2-c120.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-2p5d-tp2-tp2-c120.yaml deleted file mode 100644 index f980e5f3fa..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-2p5d-tp2-tp2-c120.yaml +++ /dev/null @@ -1,179 +0,0 @@ -schema: 2 -name: "minimax-m3-vllm-disagg-gb300-2p5d-tp2-tp2-c120-fp4-agentic" - -model: - path: "nvidia/MiniMax-M3-NVFP4" - container: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - precision: "fp4" - -identity: - model: - repo: "nvidia/MiniMax-M3-NVFP4" - container: - image: "vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7" - frameworks: - dynamo: "1.4.0.dev20260730" - -dynamo: - install: true - - source: - pypi: "1.4.0.dev20260730" -health_check: - max_attempts: 2160 - interval_seconds: 10 - -resources: - gpu_type: "gb300" - gpus_per_node: 4 - het_jobs: false - spread_workers: true -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 - - name: mooncake-master - type: mooncake-master - options: - store_config: - metadata_server: "P2PHANDSHAKE" - global_segment_size: "200GB" - local_buffer_size: "4GB" - protocol: "rdma" - device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - mode: "embedded" - enable_offload: true -environment: - PYTHONHASHSEED: "0" - -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 4 - nginx_session_affinity: true - nginx_session_affinity_header: "X-Dynamo-Session-ID" - args: - router-mode: "least-loaded" - router-session-affinity-ttl-secs: 1800 - env: - DYN_TOKENIZER: "fastokens" - DYN_TOKENIZER_CACHE_BYTES: "8589934592" - DYN_TCP_CONNECT_TIMEOUT: "120" - -engine: - type: vllm - connector: - dp_launch_mode: per_node -roles: - prefill: - nodes: 2 - workers: 2 - gpus: 2 - env: &worker_environment - HF_HUB_CACHE: "/hf_hub_cache" - HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" - TRANSFORMERS_CACHE: "/hf_hub_cache" - VLLM_ENGINE_READY_TIMEOUT_S: "3600" - DYN_TCP_CONNECT_TIMEOUT: "120" - VLLM_FLOAT32_MATMUL_PRECISION: "high" - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "trtllm" - VLLM_USE_NCCL_SYMM_MEM: "0" - VLLM_ALLREDUCE_USE_SYMM_MEM: "0" - VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" - UCX_MEMTYPE_CACHE: "n" - UCX_CUDA_IPC_ENABLE_MNNVL: "y" - UCX_MODULE_DIR: "/usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx" - UCX_RNDV_PIPELINE_ERROR_HANDLING: "y" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 2 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":true}}]}}' - enable-cumem-allocator: true - decode: - nodes: 5 - workers: 5 - gpus: 2 - - env: - <<: *worker_environment - VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl" - - MC_ENABLE_DEST_DEVICE_AFFINITY: "1" - MC_STORE_CLIENT_METRIC: "1" - MC_STORE_CLIENT_METRIC_INTERVAL: "5" - MC_TE_METRIC: "0" - args: - served-model-name: "nvidia/MiniMax-M3-NVFP4" - kv-cache-dtype: "fp8" - block-size: 128 - trust-remote-code: true - enable-prefix-caching: true - language-model-only: true - no-enable-flashinfer-autotune: true - reasoning-parser: "minimax_m3" - dyn-tool-call-parser: "minimax_m3" - dyn-reasoning-parser: "minimax_m3" - dyn-default-thinking-mode: "enabled" - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 - max-num-seqs: 1024 - stream-interval: 20 - gpu-memory-utilization: 0.9 - tensor-parallel-size: 2 - enable-expert-parallel: false - attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' - kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' - enable-cumem-allocator: true -sbatch_directives: - cpus-per-task: "72" - mem: "0" - -srun_options: - container-remap-root: "" - -benchmark: - type: custom - command: "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh" - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_EXTRA_INPUTS: "thinking:true" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: "1.0" - HF_HUB_CACHE: "/hf_hub_cache" - WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..20efc49fe2 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml @@ -0,0 +1,421 @@ +# srt-slurm recipes for minimaxm3/vllm/gb300-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: nvidia/MiniMax-M3-NVFP4 + container: vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7 + precision: fp4 + identity: + model: + repo: nvidia/MiniMax-M3-NVFP4 + container: + image: vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7 + frameworks: + dynamo: 1.4.0.dev20260730 + dynamo: + install: true + source: + pypi: 1.4.0.dev20260730 + health_check: + max_attempts: 2160 + interval_seconds: 10 + resources: + gpu_type: gb300 + gpus_per_node: 4 + het_jobs: false + services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: P2PHANDSHAKE + global_segment_size: 200GB + local_buffer_size: 4GB + protocol: rdma + device_name: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + mode: embedded + enable_offload: true + environment: + PYTHONHASHSEED: '0' + frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + args: + router-mode: least-loaded + router-session-affinity-ttl-secs: 1800 + env: + DYN_TOKENIZER: fastokens + DYN_TOKENIZER_CACHE_BYTES: '8589934592' + DYN_TCP_CONNECT_TIMEOUT: '120' + engine: + type: vllm + connector: null + dp_launch_mode: per_node + roles: + prefill: + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + DYN_TCP_CONNECT_TIMEOUT: '120' + VLLM_FLOAT32_MATMUL_PRECISION: high + VLLM_FLASHINFER_ALLREDUCE_BACKEND: trtllm + VLLM_USE_NCCL_SYMM_MEM: '0' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + UCX_MEMTYPE_CACHE: n + UCX_CUDA_IPC_ENABLE_MNNVL: y + UCX_MODULE_DIR: /usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx + UCX_RNDV_PIPELINE_ERROR_HANDLING: y + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + served-model-name: nvidia/MiniMax-M3-NVFP4 + kv-cache-dtype: fp8 + block-size: 128 + trust-remote-code: true + enable-prefix-caching: true + language-model-only: true + no-enable-flashinfer-autotune: true + reasoning-parser: minimax_m3 + dyn-tool-call-parser: minimax_m3 + dyn-reasoning-parser: minimax_m3 + max-cudagraph-capture-size: 512 + max-num-batched-tokens: 16384 + stream-interval: 20 + gpu-memory-utilization: 0.9 + attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"lookup_async":true}}]}}' + enable-cumem-allocator: true + decode: + env: + HF_HUB_CACHE: /hf_hub_cache + HUGGINGFACE_HUB_CACHE: /hf_hub_cache + TRANSFORMERS_CACHE: /hf_hub_cache + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + DYN_TCP_CONNECT_TIMEOUT: '120' + VLLM_FLOAT32_MATMUL_PRECISION: high + VLLM_FLASHINFER_ALLREDUCE_BACKEND: mnnvl + VLLM_USE_NCCL_SYMM_MEM: '0' + VLLM_ALLREDUCE_USE_SYMM_MEM: '0' + VLLM_MOONCAKE_LOAD_RECV_THREADS: '20' + UCX_MEMTYPE_CACHE: n + UCX_CUDA_IPC_ENABLE_MNNVL: y + UCX_MODULE_DIR: /usr/local/lib/python3.12/dist-packages/nixl_cu13.libs/ucx + UCX_RNDV_PIPELINE_ERROR_HANDLING: y + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NCCL_IB_HCA: mlx5_0,mlx5_1,mlx5_2,mlx5_3 + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_TE_METRIC: '0' + args: + served-model-name: nvidia/MiniMax-M3-NVFP4 + kv-cache-dtype: fp8 + block-size: 128 + trust-remote-code: true + enable-prefix-caching: true + language-model-only: true + no-enable-flashinfer-autotune: true + reasoning-parser: minimax_m3 + dyn-tool-call-parser: minimax_m3 + dyn-reasoning-parser: minimax_m3 + dyn-default-thinking-mode: enabled + max-cudagraph-capture-size: 512 + max-num-batched-tokens: 16384 + max-num-seqs: 1024 + stream-interval: 20 + gpu-memory-utilization: 0.9 + enable-expert-parallel: false + attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8","minimax_m3_msa_decode_backend":"cutlass"}' + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_connector_extra_config":{"backends":["UCX"],"enforce_handshake_compat":false,"read_validation_timeout":30.0}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"enable_lookup":false}}]}}' + enable-cumem-allocator: true + sbatch_directives: + cpus-per-task: '72' + mem: '0' + srun_options: + container-remap-root: '' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_EXTRA_INPUTS: thinking:true + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + AIPERF_SERVER_METRICS_COLLECTION_INTERVAL: '1.0' + HF_HUB_CACHE: /hf_hub_cache + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + +override_disagg_1p1d_tep4_tp4_c1_eval: + name: minimax-m3-vllm-disagg-gb300-1p1d-tep4-tp4-c1-fp4-eval-agentic + resources: + spread_workers: false + frontend: + enable_multiple_frontends: false + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + args: + tensor-parallel-size: 4 + enable-expert-parallel: true + all2all-backend: flashinfer_nvlink_one_sided + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' + decode: + nodes: 1 + workers: 1 + gpus: 4 + args: + tensor-parallel-size: 4 + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' + +override_disagg_1p1d_tep4_tp4_c1: + name: minimax-m3-vllm-disagg-gb300-1p1d-tep4-tp4-c1-fp4-agentic + resources: + spread_workers: false + frontend: + enable_multiple_frontends: false + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + args: + tensor-parallel-size: 4 + enable-expert-parallel: true + all2all-backend: flashinfer_nvlink_one_sided + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + decode: + nodes: 1 + workers: 1 + gpus: 4 + args: + tensor-parallel-size: 4 + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + +override_disagg_1p1d_tp2_tp4_c20_c24_eval: + name: minimax-m3-vllm-disagg-gb300-1p1d-tp2-tp4-c20-c24-fp4-eval-agentic + resources: + spread_workers: false + frontend: + enable_multiple_frontends: true + num_additional_frontends: 1 + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 2 + args: + tensor-parallel-size: 2 + enable-expert-parallel: false + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' + decode: + nodes: 1 + workers: 1 + gpus: 4 + args: + tensor-parallel-size: 4 + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' + +override_disagg_1p1d_tp2_tp4_c20_c24: + name: minimax-m3-vllm-disagg-gb300-1p1d-tp2-tp4-c20-c24-fp4-agentic + resources: + spread_workers: false + frontend: + enable_multiple_frontends: true + num_additional_frontends: 1 + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 2 + args: + tensor-parallel-size: 2 + enable-expert-parallel: false + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + decode: + nodes: 1 + workers: 1 + gpus: 4 + args: + tensor-parallel-size: 4 + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + +override_disagg_1p3d_dep4_tp4_c24_eval: + name: minimax-m3-vllm-disagg-gb300-1p3d-dep4-tp4-c24-fp4-eval-agentic + resources: + spread_workers: false + frontend: + enable_multiple_frontends: true + num_additional_frontends: 1 + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + args: + tensor-parallel-size: 1 + enable-expert-parallel: true + all2all-backend: flashinfer_nvlink_one_sided + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' + data-parallel-size: 4 + data-parallel-rpc-port: 13345 + decode: + nodes: 3 + workers: 3 + gpus: 4 + args: + tensor-parallel-size: 4 + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' + +override_disagg_1p3d_dep4_tp4_c24: + name: minimax-m3-vllm-disagg-gb300-1p3d-dep4-tp4-c24-fp4-agentic + resources: + spread_workers: false + frontend: + enable_multiple_frontends: true + num_additional_frontends: 1 + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + args: + tensor-parallel-size: 1 + enable-expert-parallel: true + all2all-backend: flashinfer_nvlink_one_sided + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + data-parallel-size: 4 + data-parallel-rpc-port: 13345 + decode: + nodes: 3 + workers: 3 + gpus: 4 + args: + tensor-parallel-size: 4 + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + +override_disagg_1p3d_tp2_tp2_c48_eval: + name: minimax-m3-vllm-disagg-gb300-1p3d-tp2-tp2-c48-fp4-eval-agentic + resources: + spread_workers: true + frontend: + enable_multiple_frontends: true + num_additional_frontends: 4 + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 2 + args: + tensor-parallel-size: 2 + enable-expert-parallel: false + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' + decode: + nodes: 3 + workers: 3 + gpus: 2 + args: + tensor-parallel-size: 2 + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' + +override_disagg_1p3d_tp2_tp2_c48: + name: minimax-m3-vllm-disagg-gb300-1p3d-tp2-tp2-c48-fp4-agentic + resources: + spread_workers: true + frontend: + enable_multiple_frontends: true + num_additional_frontends: 4 + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 2 + args: + tensor-parallel-size: 2 + enable-expert-parallel: false + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + decode: + nodes: 3 + workers: 3 + gpus: 2 + args: + tensor-parallel-size: 2 + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + +override_disagg_2p5d_tp2_tp2_c120_eval: + name: minimax-m3-vllm-disagg-gb300-2p5d-tp2-tp2-c120-fp4-eval-agentic + resources: + spread_workers: true + frontend: + enable_multiple_frontends: true + num_additional_frontends: 4 + roles: + prefill: + nodes: 2 + workers: 2 + gpus: 2 + args: + tensor-parallel-size: 2 + enable-expert-parallel: false + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' + decode: + nodes: 5 + workers: 5 + gpus: 2 + args: + tensor-parallel-size: 2 + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' + +override_disagg_2p5d_tp2_tp2_c120: + name: minimax-m3-vllm-disagg-gb300-2p5d-tp2-tp2-c120-fp4-agentic + resources: + spread_workers: true + frontend: + enable_multiple_frontends: true + num_additional_frontends: 4 + roles: + prefill: + nodes: 2 + workers: 2 + gpus: 2 + args: + tensor-parallel-size: 2 + enable-expert-parallel: false + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' + decode: + nodes: 5 + workers: 5 + gpus: 2 + args: + tensor-parallel-size: 2 + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN","rejection_sample_method":"block"}' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c16-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c16-mtp.yaml deleted file mode 100644 index a3f8bb681c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c16-mtp.yaml +++ /dev/null @@ -1,175 +0,0 @@ -# Experimental; retain only measured improvements over the current Pareto frontier. -schema: 2 -name: qwen35-b200-fp8-agentx-disagg-1p1d-p-tp4-d-tp4-hicache-c16-mtp -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -engine: sglang -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -resources: - gpu_type: b200 - gpus_per_node: 8 -slurm: - time_limit: '4:00:00' -frontend: - type: dynamo - nginx_container: nginx-sqsh - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - num_additional_frontends: 1 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - SGLANG_ENABLE_SPEC_V2: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - NCCL_NVLS_ENABLE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - PYTHONNOUSERSITE: '1' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - trust-remote-code: true - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - page-size: 64 - mem-fraction-static: 0.8 - context-length: 262144 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - max-running-requests: 32 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - enable-metrics: true - enable-cache-report: true - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - disable-cuda-graph: true - enable-symm-mem: true - scheduler-recv-interval: 10 - stream-interval: 50 - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - enable-hierarchical-cache: true - hicache-size: 104 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through_selective - decode: - nodes: colocate - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - SGLANG_ENABLE_SPEC_V2: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - NCCL_NVLS_ENABLE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - PYTHONNOUSERSITE: '1' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - trust-remote-code: true - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - page-size: 64 - mem-fraction-static: 0.8 - context-length: 262144 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - max-running-requests: 32 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - enable-metrics: true - enable-cache-report: true - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs-decode: 16 - stream-interval: 50 - enable-symm-mem: true - scheduler-recv-interval: 10 - mamba-radix-cache-strategy: no_buffer - mamba-track-interval: 128 -health_check: - max_attempts: 720 - interval_seconds: 10 -sbatch_directives: - mem: '0' -srun_options: - mem: '0' - container-remap-root: '' -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - concurrencies: - - 16 - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - SRT_MEASUREMENT_WINDOW_BENCHMARK_TYPE: custom - SRT_MEASUREMENT_WINDOW_CONCURRENCIES: '16' - SRT_MEASUREMENT_WINDOW_RESULT_ROOT: /logs - PORT: '8000' - IS_MULTINODE: 'true' - EVAL_FRAMEWORK: lm-eval - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_MAX_CONTEXT_LENGTH: '262144' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c24-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c24-mtp.yaml deleted file mode 100644 index a9af6f1251..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c24-mtp.yaml +++ /dev/null @@ -1,175 +0,0 @@ -# Experimental; retain only measured improvements over the current Pareto frontier. -schema: 2 -name: qwen35-b200-fp8-agentx-disagg-1p1d-p-tp4-d-tp4-hicache-c24-mtp -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -engine: sglang -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -resources: - gpu_type: b200 - gpus_per_node: 8 -slurm: - time_limit: '4:00:00' -frontend: - type: dynamo - nginx_container: nginx-sqsh - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - num_additional_frontends: 1 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - SGLANG_ENABLE_SPEC_V2: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - NCCL_NVLS_ENABLE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - PYTHONNOUSERSITE: '1' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - trust-remote-code: true - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - page-size: 64 - mem-fraction-static: 0.8 - context-length: 262144 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - max-running-requests: 48 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - enable-metrics: true - enable-cache-report: true - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - disable-cuda-graph: true - enable-symm-mem: true - scheduler-recv-interval: 10 - stream-interval: 50 - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - enable-hierarchical-cache: true - hicache-size: 104 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through_selective - decode: - nodes: colocate - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - SGLANG_ENABLE_SPEC_V2: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - NCCL_NVLS_ENABLE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - PYTHONNOUSERSITE: '1' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - trust-remote-code: true - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - page-size: 64 - mem-fraction-static: 0.8 - context-length: 262144 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - max-running-requests: 48 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - enable-metrics: true - enable-cache-report: true - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs-decode: 24 - stream-interval: 50 - enable-symm-mem: true - scheduler-recv-interval: 10 - mamba-radix-cache-strategy: no_buffer - mamba-track-interval: 128 -health_check: - max_attempts: 720 - interval_seconds: 10 -sbatch_directives: - mem: '0' -srun_options: - mem: '0' - container-remap-root: '' -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - concurrencies: - - 24 - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - SRT_MEASUREMENT_WINDOW_BENCHMARK_TYPE: custom - SRT_MEASUREMENT_WINDOW_CONCURRENCIES: '24' - SRT_MEASUREMENT_WINDOW_RESULT_ROOT: /logs - PORT: '8000' - IS_MULTINODE: 'true' - EVAL_FRAMEWORK: lm-eval - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_MAX_CONTEXT_LENGTH: '262144' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c32-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c32-mtp.yaml deleted file mode 100644 index 5dab4917a0..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c32-mtp.yaml +++ /dev/null @@ -1,175 +0,0 @@ -# Fast replay improves the published frontier; canonical sweep qualification is pending. -schema: 2 -name: qwen35-b200-fp8-agentx-disagg-1p1d-p-tp4-d-tp4-hicache-c32-mtp -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -engine: sglang -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -resources: - gpu_type: b200 - gpus_per_node: 8 -slurm: - time_limit: '4:00:00' -frontend: - type: dynamo - nginx_container: nginx-sqsh - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - num_additional_frontends: 1 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - SGLANG_ENABLE_SPEC_V2: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - NCCL_NVLS_ENABLE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - PYTHONNOUSERSITE: '1' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - trust-remote-code: true - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - page-size: 64 - mem-fraction-static: 0.8 - context-length: 262144 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - max-running-requests: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - enable-metrics: true - enable-cache-report: true - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - disable-cuda-graph: true - enable-symm-mem: true - scheduler-recv-interval: 10 - stream-interval: 50 - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - enable-hierarchical-cache: true - hicache-size: 104 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through_selective - decode: - nodes: colocate - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - SGLANG_ENABLE_SPEC_V2: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - NCCL_NVLS_ENABLE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - PYTHONNOUSERSITE: '1' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - trust-remote-code: true - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - page-size: 64 - mem-fraction-static: 0.8 - context-length: 262144 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - max-running-requests: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - enable-metrics: true - enable-cache-report: true - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs-decode: 32 - stream-interval: 50 - enable-symm-mem: true - scheduler-recv-interval: 10 - mamba-radix-cache-strategy: no_buffer - mamba-track-interval: 128 -health_check: - max_attempts: 720 - interval_seconds: 10 -sbatch_directives: - mem: '0' -srun_options: - mem: '0' - container-remap-root: '' -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - concurrencies: - - 32 - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - SRT_MEASUREMENT_WINDOW_BENCHMARK_TYPE: custom - SRT_MEASUREMENT_WINDOW_CONCURRENCIES: '32' - SRT_MEASUREMENT_WINDOW_RESULT_ROOT: /logs - PORT: '8000' - IS_MULTINODE: 'true' - EVAL_FRAMEWORK: lm-eval - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_MAX_CONTEXT_LENGTH: '262144' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c48-write-through-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c48-write-through-mtp.yaml deleted file mode 100644 index 5197d3be5a..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c48-write-through-mtp.yaml +++ /dev/null @@ -1,176 +0,0 @@ -# Experimental; retain only measured improvements over the current Pareto frontier. -schema: 2 -name: qwen35-b200-fp8-agentx-disagg-1p1d-p-tp4-d-tp4-hicache-c48-write-through-mtp -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -engine: sglang -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -resources: - gpu_type: b200 - gpus_per_node: 8 -slurm: - time_limit: '4:00:00' -frontend: - type: dynamo - nginx_container: nginx-sqsh - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - num_additional_frontends: 1 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - SGLANG_ENABLE_SPEC_V2: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - NCCL_NVLS_ENABLE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - PYTHONNOUSERSITE: '1' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - trust-remote-code: true - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - page-size: 64 - mem-fraction-static: 0.8 - context-length: 262144 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - max-running-requests: 96 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - enable-metrics: true - enable-cache-report: true - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - disable-cuda-graph: true - enable-symm-mem: true - scheduler-recv-interval: 10 - stream-interval: 50 - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - enable-hierarchical-cache: true - # Combined KV/Mamba budget: 860 GB per TP4 prefill worker. - hicache-size: 215 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through - decode: - nodes: colocate - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - SGLANG_ENABLE_SPEC_V2: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - NCCL_NVLS_ENABLE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - PYTHONNOUSERSITE: '1' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - trust-remote-code: true - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - page-size: 64 - mem-fraction-static: 0.88 - context-length: 262144 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - max-running-requests: 96 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - enable-metrics: true - enable-cache-report: true - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs-decode: 48 - stream-interval: 50 - enable-symm-mem: true - scheduler-recv-interval: 10 - mamba-radix-cache-strategy: no_buffer - mamba-track-interval: 128 -health_check: - max_attempts: 720 - interval_seconds: 10 -sbatch_directives: - mem: '0' -srun_options: - mem: '0' - container-remap-root: '' -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - concurrencies: - - 48 - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - SRT_MEASUREMENT_WINDOW_BENCHMARK_TYPE: custom - SRT_MEASUREMENT_WINDOW_CONCURRENCIES: '48' - SRT_MEASUREMENT_WINDOW_RESULT_ROOT: /logs - PORT: '8000' - IS_MULTINODE: 'true' - EVAL_FRAMEWORK: lm-eval - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_MAX_CONTEXT_LENGTH: '262144' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c64-write-through-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c64-write-through-mtp.yaml deleted file mode 100644 index ac6a00a1d8..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c64-write-through-mtp.yaml +++ /dev/null @@ -1,176 +0,0 @@ -# Short replay improves the published frontier; full official qualification remains required. -schema: 2 -name: qwen35-b200-fp8-agentx-disagg-1p1d-p-tp4-d-tp4-hicache-c64-write-through-mtp -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -engine: sglang -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -resources: - gpu_type: b200 - gpus_per_node: 8 -slurm: - time_limit: '4:00:00' -frontend: - type: dynamo - nginx_container: nginx-sqsh - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - num_additional_frontends: 1 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - SGLANG_ENABLE_SPEC_V2: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - NCCL_NVLS_ENABLE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - PYTHONNOUSERSITE: '1' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - trust-remote-code: true - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - page-size: 64 - mem-fraction-static: 0.8 - context-length: 262144 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - max-running-requests: 128 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - enable-metrics: true - enable-cache-report: true - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - disable-cuda-graph: true - enable-symm-mem: true - scheduler-recv-interval: 10 - stream-interval: 50 - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - enable-hierarchical-cache: true - # Combined KV/Mamba budget: 860 GB per TP4 prefill worker. - hicache-size: 215 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through - decode: - nodes: colocate - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - SGLANG_ENABLE_SPEC_V2: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - NCCL_NVLS_ENABLE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - PYTHONNOUSERSITE: '1' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - trust-remote-code: true - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - page-size: 64 - mem-fraction-static: 0.88 - context-length: 262144 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - max-running-requests: 128 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - enable-metrics: true - enable-cache-report: true - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs-decode: 64 - stream-interval: 50 - enable-symm-mem: true - scheduler-recv-interval: 10 - mamba-radix-cache-strategy: no_buffer - mamba-track-interval: 128 -health_check: - max_attempts: 720 - interval_seconds: 10 -sbatch_directives: - mem: '0' -srun_options: - mem: '0' - container-remap-root: '' -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - concurrencies: - - 64 - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - SRT_MEASUREMENT_WINDOW_BENCHMARK_TYPE: custom - SRT_MEASUREMENT_WINDOW_CONCURRENCIES: '64' - SRT_MEASUREMENT_WINDOW_RESULT_ROOT: /logs - PORT: '8000' - IS_MULTINODE: 'true' - EVAL_FRAMEWORK: lm-eval - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_MAX_CONTEXT_LENGTH: '262144' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c8-mtp.yaml deleted file mode 100644 index 2492edcae4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c8-mtp.yaml +++ /dev/null @@ -1,175 +0,0 @@ -# Experimental; retain only measured improvements over the current Pareto frontier. -schema: 2 -name: qwen35-b200-fp8-agentx-disagg-1p1d-p-tp4-d-tp4-hicache-c8-mtp -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -engine: sglang -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -resources: - gpu_type: b200 - gpus_per_node: 8 -slurm: - time_limit: '4:00:00' -frontend: - type: dynamo - nginx_container: nginx-sqsh - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - num_additional_frontends: 1 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - SGLANG_ENABLE_SPEC_V2: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - NCCL_NVLS_ENABLE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - PYTHONNOUSERSITE: '1' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - trust-remote-code: true - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - page-size: 64 - mem-fraction-static: 0.8 - context-length: 262144 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - max-running-requests: 32 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - enable-metrics: true - enable-cache-report: true - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - disable-cuda-graph: true - enable-symm-mem: true - scheduler-recv-interval: 10 - stream-interval: 50 - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - enable-hierarchical-cache: true - hicache-size: 104 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through_selective - decode: - nodes: colocate - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - SGLANG_ENABLE_SPEC_V2: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - NCCL_NVLS_ENABLE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - PYTHONNOUSERSITE: '1' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - trust-remote-code: true - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - page-size: 64 - mem-fraction-static: 0.8 - context-length: 262144 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - max-running-requests: 32 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - enable-metrics: true - enable-cache-report: true - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs-decode: 8 - stream-interval: 50 - enable-symm-mem: true - scheduler-recv-interval: 10 - mamba-radix-cache-strategy: no_buffer - mamba-track-interval: 128 -health_check: - max_attempts: 720 - interval_seconds: 10 -sbatch_directives: - mem: '0' -srun_options: - mem: '0' - container-remap-root: '' -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - collector_join_timeout_seconds: 12 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - concurrencies: - - 8 - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - SRT_MEASUREMENT_WINDOW_BENCHMARK_TYPE: custom - SRT_MEASUREMENT_WINDOW_CONCURRENCIES: '8' - SRT_MEASUREMENT_WINDOW_RESULT_ROOT: /logs - PORT: '8000' - IS_MULTINODE: 'true' - EVAL_FRAMEWORK: lm-eval - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_MAX_CONTEXT_LENGTH: '262144' - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml new file mode 100644 index 0000000000..e37dc348c4 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml @@ -0,0 +1,286 @@ +# srt-slurm recipes for qwen3.5/sglang/b200-fp8/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: qwen3.5-fp8 + container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 + precision: fp8 + engine: sglang + dynamo: + install: true + source: + rev: 805a77f053d807b0d8def5d27f674a6df0ed839e + resources: + gpu_type: b200 + gpus_per_node: 8 + slurm: + time_limit: '4:00:00' + frontend: + type: dynamo + nginx_container: nginx-sqsh + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: true + num_additional_frontends: 1 + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + PYTHONUNBUFFERED: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com + SGLANG_ENABLE_SPEC_V2: '1' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + SGLANG_TIMEOUT_KEEP_ALIVE: '1800' + NCCL_NVLS_ENABLE: '1' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + PYTHONNOUSERSITE: '1' + MC_INTRANODE_NVLINK: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK + args: + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + trust-remote-code: true + tensor-parallel-size: 4 + expert-parallel-size: 1 + data-parallel-size: 1 + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + page-size: 64 + mem-fraction-static: 0.8 + context-length: 262144 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + enable-metrics: true + enable-cache-report: true + disaggregation-mode: prefill + disaggregation-transfer-backend: mooncake + disable-cuda-graph: true + enable-symm-mem: true + scheduler-recv-interval: 10 + stream-interval: 50 + mamba-radix-cache-strategy: extra_buffer + mamba-track-interval: 2048 + enable-hierarchical-cache: true + hicache-io-backend: kernel + hicache-mem-layout: page_first + decode: + nodes: colocate + workers: 1 + gpus: 4 + env: + PYTHONUNBUFFERED: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com + SGLANG_ENABLE_SPEC_V2: '1' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + SGLANG_TIMEOUT_KEEP_ALIVE: '1800' + NCCL_NVLS_ENABLE: '1' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + PYTHONNOUSERSITE: '1' + MC_INTRANODE_NVLINK: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK + args: + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + trust-remote-code: true + tensor-parallel-size: 4 + expert-parallel-size: 1 + data-parallel-size: 1 + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + page-size: 64 + context-length: 262144 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + enable-metrics: true + enable-cache-report: true + disaggregation-mode: decode + disaggregation-transfer-backend: mooncake + disable-radix-cache: true + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + stream-interval: 50 + enable-symm-mem: true + scheduler-recv-interval: 10 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 128 + health_check: + max_attempts: 720 + interval_seconds: 10 + sbatch_directives: + mem: '0' + srun_options: + mem: '0' + container-remap-root: '' + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + collector_join_timeout_seconds: 12 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + SRT_MEASUREMENT_WINDOW_BENCHMARK_TYPE: custom + SRT_MEASUREMENT_WINDOW_RESULT_ROOT: /logs + PORT: '8000' + IS_MULTINODE: 'true' + EVAL_FRAMEWORK: lm-eval + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k + AIPERF_MAX_CONTEXT_LENGTH: '262144' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' + +# Experimental; retain only measured improvements over the current Pareto frontier. +override_disagg_1p1d_p_tp4_d_tp4_hicache_c16_mtp: + name: qwen35-b200-fp8-agentx-disagg-1p1d-p-tp4-d-tp4-hicache-c16-mtp + roles: + prefill: + args: + max-running-requests: 32 + hicache-size: 104 + hicache-write-policy: write_through_selective + decode: + args: + mem-fraction-static: 0.8 + max-running-requests: 32 + cuda-graph-max-bs-decode: 16 + benchmark: + concurrencies: [16] + env: + SRT_MEASUREMENT_WINDOW_CONCURRENCIES: '16' + +# Experimental; retain only measured improvements over the current Pareto frontier. +override_disagg_1p1d_p_tp4_d_tp4_hicache_c24_mtp: + name: qwen35-b200-fp8-agentx-disagg-1p1d-p-tp4-d-tp4-hicache-c24-mtp + roles: + prefill: + args: + max-running-requests: 48 + hicache-size: 104 + hicache-write-policy: write_through_selective + decode: + args: + mem-fraction-static: 0.8 + max-running-requests: 48 + cuda-graph-max-bs-decode: 24 + benchmark: + concurrencies: [24] + env: + SRT_MEASUREMENT_WINDOW_CONCURRENCIES: '24' + +# Fast replay improves the published frontier; canonical sweep qualification is pending. +override_disagg_1p1d_p_tp4_d_tp4_hicache_c32_mtp: + name: qwen35-b200-fp8-agentx-disagg-1p1d-p-tp4-d-tp4-hicache-c32-mtp + roles: + prefill: + args: + max-running-requests: 64 + hicache-size: 104 + hicache-write-policy: write_through_selective + decode: + args: + mem-fraction-static: 0.8 + max-running-requests: 64 + cuda-graph-max-bs-decode: 32 + benchmark: + concurrencies: [32] + env: + SRT_MEASUREMENT_WINDOW_CONCURRENCIES: '32' + +# Experimental; retain only measured improvements over the current Pareto frontier. +override_disagg_1p1d_p_tp4_d_tp4_hicache_c48_write_through_mtp: + name: qwen35-b200-fp8-agentx-disagg-1p1d-p-tp4-d-tp4-hicache-c48-write-through-mtp + roles: + prefill: + args: + max-running-requests: 96 + # Combined KV/Mamba budget: 860 GB per TP4 prefill worker. + hicache-size: 215 + hicache-write-policy: write_through + decode: + args: + mem-fraction-static: 0.88 + max-running-requests: 96 + cuda-graph-max-bs-decode: 48 + benchmark: + concurrencies: [48] + env: + SRT_MEASUREMENT_WINDOW_CONCURRENCIES: '48' + +# Short replay improves the published frontier; full official qualification remains required. +override_disagg_1p1d_p_tp4_d_tp4_hicache_c64_write_through_mtp: + name: qwen35-b200-fp8-agentx-disagg-1p1d-p-tp4-d-tp4-hicache-c64-write-through-mtp + roles: + prefill: + args: + max-running-requests: 128 + # Combined KV/Mamba budget: 860 GB per TP4 prefill worker. + hicache-size: 215 + hicache-write-policy: write_through + decode: + args: + mem-fraction-static: 0.88 + max-running-requests: 128 + cuda-graph-max-bs-decode: 64 + benchmark: + concurrencies: [64] + env: + SRT_MEASUREMENT_WINDOW_CONCURRENCIES: '64' + +# Experimental; retain only measured improvements over the current Pareto frontier. +override_disagg_1p1d_p_tp4_d_tp4_hicache_c8_mtp: + name: qwen35-b200-fp8-agentx-disagg-1p1d-p-tp4-d-tp4-hicache-c8-mtp + roles: + prefill: + args: + max-running-requests: 32 + hicache-size: 104 + hicache-write-policy: write_through_selective + decode: + args: + mem-fraction-static: 0.8 + max-running-requests: 32 + cuda-graph-max-bs-decode: 8 + benchmark: + concurrencies: [8] + env: + SRT_MEASUREMENT_WINDOW_CONCURRENCIES: '8' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c32-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c32-mtp-hicache.yaml deleted file mode 100644 index c73b3f0f1f..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c32-mtp-hicache.yaml +++ /dev/null @@ -1,179 +0,0 @@ -# Colocated TP2+EP2 prefill and decode with write-through HiCache. -schema: 2 -name: qwen35-b300-disagg-1p1d-tp2ep2-tp2ep2-colocated-c32-mtp-hicache -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -identity: - model: - repo: Qwen/Qwen3.5-397B-A17B-FP8 - revision: ea5b4f81096f3901c91dea97f81324302495781d - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -slurm: - time_limit: '4:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: b300 - gpus_per_node: 8 -frontend: - type: dynamo - nginx_session_affinity: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 2 - expert-parallel-size: 2 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - disable-cuda-graph: true - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 64 - enable-hierarchical-cache: true - hicache-size: 298 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through - decode: - nodes: colocate - workers: 1 - gpus: 2 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 2 - expert-parallel-size: 2 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.88 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - mamba-radix-cache-strategy: no_buffer - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 64 - cuda-graph-max-bs-decode: 32 - disaggregation-decode-extra-slots: 0 - disaggregation-decode-retraction-backup: cpu_tensor -sbatch_directives: - mem: '0' - cpus-per-task: '144' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - ENABLE_AGENTX_POWER: '0' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - concurrencies: - - 32 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c40-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c40-mtp-hicache.yaml deleted file mode 100644 index 0150c174ba..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c40-mtp-hicache.yaml +++ /dev/null @@ -1,179 +0,0 @@ -# Colocated TP2+EP2 prefill and decode with write-through HiCache. -schema: 2 -name: qwen35-b300-disagg-1p1d-tp2ep2-tp2ep2-colocated-c40-mtp-hicache -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -identity: - model: - repo: Qwen/Qwen3.5-397B-A17B-FP8 - revision: ea5b4f81096f3901c91dea97f81324302495781d - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -slurm: - time_limit: '4:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: b300 - gpus_per_node: 8 -frontend: - type: dynamo - nginx_session_affinity: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 2 - expert-parallel-size: 2 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - disable-cuda-graph: true - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 64 - enable-hierarchical-cache: true - hicache-size: 298 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through - decode: - nodes: colocate - workers: 1 - gpus: 2 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 2 - expert-parallel-size: 2 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.88 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - mamba-radix-cache-strategy: no_buffer - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 64 - cuda-graph-max-bs-decode: 48 - disaggregation-decode-extra-slots: 0 - disaggregation-decode-retraction-backup: cpu_tensor -sbatch_directives: - mem: '0' - cpus-per-task: '144' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - ENABLE_AGENTX_POWER: '0' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - concurrencies: - - 40 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c44-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c44-mtp-hicache.yaml deleted file mode 100644 index cecb894de7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c44-mtp-hicache.yaml +++ /dev/null @@ -1,179 +0,0 @@ -# Colocated TP2+EP2 prefill and decode with write-through HiCache. -schema: 2 -name: qwen35-b300-disagg-1p1d-tp2ep2-tp2ep2-colocated-c44-mtp-hicache -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -identity: - model: - repo: Qwen/Qwen3.5-397B-A17B-FP8 - revision: ea5b4f81096f3901c91dea97f81324302495781d - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -slurm: - time_limit: '4:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: b300 - gpus_per_node: 8 -frontend: - type: dynamo - nginx_session_affinity: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 2 - expert-parallel-size: 2 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - disable-cuda-graph: true - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 64 - enable-hierarchical-cache: true - hicache-size: 298 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through - decode: - nodes: colocate - workers: 1 - gpus: 2 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 2 - expert-parallel-size: 2 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.88 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - mamba-radix-cache-strategy: no_buffer - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 64 - cuda-graph-max-bs-decode: 48 - disaggregation-decode-extra-slots: 0 - disaggregation-decode-retraction-backup: cpu_tensor -sbatch_directives: - mem: '0' - cpus-per-task: '144' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - ENABLE_AGENTX_POWER: '0' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - concurrencies: - - 44 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c48-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c48-mtp-hicache.yaml deleted file mode 100644 index d5b99fbbcd..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c48-mtp-hicache.yaml +++ /dev/null @@ -1,179 +0,0 @@ -# Colocated TP2+EP2 prefill and decode with write-through HiCache. -schema: 2 -name: qwen35-b300-disagg-1p1d-tp2ep2-tp2ep2-colocated-c48-mtp-hicache -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -identity: - model: - repo: Qwen/Qwen3.5-397B-A17B-FP8 - revision: ea5b4f81096f3901c91dea97f81324302495781d - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -slurm: - time_limit: '4:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: b300 - gpus_per_node: 8 -frontend: - type: dynamo - nginx_session_affinity: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 2 - expert-parallel-size: 2 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - disable-cuda-graph: true - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 64 - enable-hierarchical-cache: true - hicache-size: 298 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through - decode: - nodes: colocate - workers: 1 - gpus: 2 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 2 - expert-parallel-size: 2 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.92 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - mamba-radix-cache-strategy: no_buffer - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 64 - cuda-graph-max-bs-decode: 48 - disaggregation-decode-extra-slots: 0 - disaggregation-decode-retraction-backup: cpu_tensor -sbatch_directives: - mem: '0' - cpus-per-task: '144' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - ENABLE_AGENTX_POWER: '0' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - concurrencies: - - 48 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c56-replayssm-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c56-replayssm-mtp-hicache.yaml deleted file mode 100644 index f007aad45b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c56-replayssm-mtp-hicache.yaml +++ /dev/null @@ -1,180 +0,0 @@ -# Colocated TP2+EP2 prefill and decode with write-through HiCache. -schema: 2 -name: qwen35-b300-disagg-1p1d-tp2ep2-tp2ep2-colocated-c56-replayssm-mtp-hicache -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -identity: - model: - repo: Qwen/Qwen3.5-397B-A17B-FP8 - revision: ea5b4f81096f3901c91dea97f81324302495781d - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -slurm: - time_limit: '4:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: b300 - gpus_per_node: 8 -frontend: - type: dynamo - nginx_session_affinity: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 2 - expert-parallel-size: 2 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - disable-cuda-graph: true - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 64 - enable-hierarchical-cache: true - hicache-size: 298 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through - decode: - nodes: colocate - workers: 1 - gpus: 2 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 2 - expert-parallel-size: 2 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.92 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - mamba-radix-cache-strategy: no_buffer - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - enable-linear-replayssm-spec: true - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 64 - cuda-graph-max-bs-decode: 48 - disaggregation-decode-extra-slots: 0 - disaggregation-decode-retraction-backup: cpu_tensor -sbatch_directives: - mem: '0' - cpus-per-task: '144' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - ENABLE_AGENTX_POWER: '0' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - concurrencies: - - 56 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c12-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c12-mtp-hicache.yaml deleted file mode 100644 index a8d1f5bc62..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c12-mtp-hicache.yaml +++ /dev/null @@ -1,179 +0,0 @@ -# Colocated prefill and decode with prefill HiCache. -schema: 2 -name: qwen35-b300-disagg-1p1d-tp4-tp4-colocated-c12-mtp-hicache -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -identity: - model: - repo: Qwen/Qwen3.5-397B-A17B-FP8 - revision: ea5b4f81096f3901c91dea97f81324302495781d - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -slurm: - time_limit: '4:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: b300 - gpus_per_node: 8 -frontend: - type: dynamo - nginx_session_affinity: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - disable-cuda-graph: true - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 32 - enable-hierarchical-cache: true - hicache-size: 72 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through_selective - decode: - nodes: colocate - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - mamba-radix-cache-strategy: no_buffer - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 32 - cuda-graph-max-bs-decode: 16 - disaggregation-decode-extra-slots: 0 - disaggregation-decode-retraction-backup: cpu_tensor -sbatch_directives: - mem: '0' - cpus-per-task: '144' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - ENABLE_AGENTX_POWER: '0' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - concurrencies: - - 12 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c24-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c24-mtp-hicache.yaml deleted file mode 100644 index 1b1fe769ad..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c24-mtp-hicache.yaml +++ /dev/null @@ -1,179 +0,0 @@ -# Colocated TP4 prefill and decode with prefill HiCache. -schema: 2 -name: qwen35-b300-disagg-1p1d-tp4-tp4-colocated-c24-mtp-hicache -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -identity: - model: - repo: Qwen/Qwen3.5-397B-A17B-FP8 - revision: ea5b4f81096f3901c91dea97f81324302495781d - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -slurm: - time_limit: '4:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: b300 - gpus_per_node: 8 -frontend: - type: dynamo - nginx_session_affinity: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - disable-cuda-graph: true - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 32 - enable-hierarchical-cache: true - hicache-size: 72 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through_selective - decode: - nodes: colocate - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - mamba-radix-cache-strategy: no_buffer - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 32 - cuda-graph-max-bs-decode: 32 - disaggregation-decode-extra-slots: 0 - disaggregation-decode-retraction-backup: cpu_tensor -sbatch_directives: - mem: '0' - cpus-per-task: '144' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - ENABLE_AGENTX_POWER: '0' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - concurrencies: - - 24 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c4-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c4-mtp-hicache.yaml deleted file mode 100644 index d88c06b458..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c4-mtp-hicache.yaml +++ /dev/null @@ -1,179 +0,0 @@ -# Colocated prefill and decode with prefill HiCache. -schema: 2 -name: qwen35-b300-disagg-1p1d-tp4-tp4-colocated-c4-mtp-hicache -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -identity: - model: - repo: Qwen/Qwen3.5-397B-A17B-FP8 - revision: ea5b4f81096f3901c91dea97f81324302495781d - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -slurm: - time_limit: '4:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: b300 - gpus_per_node: 8 -frontend: - type: dynamo - nginx_session_affinity: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - disable-cuda-graph: true - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 32 - enable-hierarchical-cache: true - hicache-size: 72 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through_selective - decode: - nodes: colocate - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - mamba-radix-cache-strategy: no_buffer - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 32 - cuda-graph-max-bs-decode: 16 - disaggregation-decode-extra-slots: 0 - disaggregation-decode-retraction-backup: cpu_tensor -sbatch_directives: - mem: '0' - cpus-per-task: '144' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - ENABLE_AGENTX_POWER: '0' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - concurrencies: - - 4 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4ep4-tp4-colocated-c32-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4ep4-tp4-colocated-c32-mtp-hicache.yaml deleted file mode 100644 index ce87f0369b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4ep4-tp4-colocated-c32-mtp-hicache.yaml +++ /dev/null @@ -1,179 +0,0 @@ -# Colocated prefill and decode with prefill HiCache. -schema: 2 -name: qwen35-b300-disagg-1p1d-tp4ep4-tp4-colocated-c32-mtp-hicache -model: - path: qwen3.5-fp8 - container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 - precision: fp8 -identity: - model: - repo: Qwen/Qwen3.5-397B-A17B-FP8 - revision: ea5b4f81096f3901c91dea97f81324302495781d - container: - image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 -dynamo: - install: true - source: - rev: 805a77f053d807b0d8def5d27f674a6df0ed839e -slurm: - time_limit: '4:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: b300 - gpus_per_node: 8 -frontend: - type: dynamo - nginx_session_affinity: true - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - expert-parallel-size: 4 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: prefill - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - disable-cuda-graph: true - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 2048 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 32 - enable-hierarchical-cache: true - hicache-size: 72 - hicache-io-backend: kernel - hicache-mem-layout: page_first - hicache-write-policy: write_through_selective - decode: - nodes: colocate - workers: 1 - gpus: 4 - env: - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com - NCCL_NVLS_ENABLE: '1' - SGLANG_TIMEOUT_KEEP_ALIVE: '1800' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - MC_INTRANODE_NVLINK: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_CUDA_ARCH_LIST: '10.0' - PYTHONNOUSERSITE: '1' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - args: - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - model-path: /model/ - trust-remote-code: true - quantization: fp8 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - expert-parallel-size: 1 - data-parallel-size: 1 - mamba-ssm-dtype: bfloat16 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - enable-symm-mem: true - enable-cache-report: true - enable-metrics: true - context-length: 262144 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - page-size: 64 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - disaggregation-mode: decode - disaggregation-transfer-backend: mooncake - watchdog-timeout: 3600 - mamba-radix-cache-strategy: no_buffer - disable-radix-cache: true - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - scheduler-recv-interval: 10 - stream-interval: 50 - max-running-requests: 32 - cuda-graph-max-bs-decode: 32 - disaggregation-decode-extra-slots: 0 - disaggregation-decode-retraction-backup: cpu_tensor -sbatch_directives: - mem: '0' - cpus-per-task: '144' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - ENABLE_AGENTX_POWER: '0' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' - concurrencies: - - 32 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml new file mode 100644 index 0000000000..2bf860310a --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml @@ -0,0 +1,377 @@ +# srt-slurm recipes for qwen3.5/sglang/b300-fp8/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: qwen3.5-fp8 + container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 + precision: fp8 + identity: + model: + repo: Qwen/Qwen3.5-397B-A17B-FP8 + revision: ea5b4f81096f3901c91dea97f81324302495781d + container: + image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 + dynamo: + install: true + source: + rev: 805a77f053d807b0d8def5d27f674a6df0ed839e + slurm: + time_limit: '4:00:00' + health_check: + max_attempts: 1440 + interval_seconds: 10 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: dynamo + nginx_session_affinity: true + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + nginx_session_affinity_header: X-Dynamo-Session-ID + engine: sglang + roles: + prefill: + nodes: 1 + workers: 1 + env: + PYTHONUNBUFFERED: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com + NCCL_NVLS_ENABLE: '1' + SGLANG_TIMEOUT_KEEP_ALIVE: '1800' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + MC_INTRANODE_NVLINK: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_CUDA_ARCH_LIST: '10.0' + PYTHONNOUSERSITE: '1' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + args: + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + model-path: /model/ + trust-remote-code: true + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + data-parallel-size: 1 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-prefill-backend: flashinfer + linear-attn-decode-backend: flashinfer + enable-symm-mem: true + enable-cache-report: true + enable-metrics: true + context-length: 262144 + mem-fraction-static: 0.8 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + page-size: 64 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + disaggregation-mode: prefill + disaggregation-transfer-backend: mooncake + watchdog-timeout: 3600 + disable-cuda-graph: true + mamba-radix-cache-strategy: extra_buffer + mamba-track-interval: 2048 + scheduler-recv-interval: 10 + stream-interval: 50 + enable-hierarchical-cache: true + hicache-io-backend: kernel + hicache-mem-layout: page_first + decode: + nodes: colocate + workers: 1 + env: + PYTHONUNBUFFERED: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PIP_EXTRA_INDEX_URL: https://pypi.nvidia.com + NCCL_NVLS_ENABLE: '1' + SGLANG_TIMEOUT_KEEP_ALIVE: '1800' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + MC_INTRANODE_NVLINK: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: INTRA_NODE_NVLINK + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_CUDA_ARCH_LIST: '10.0' + PYTHONNOUSERSITE: '1' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + args: + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + model-path: /model/ + trust-remote-code: true + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + data-parallel-size: 1 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-prefill-backend: flashinfer + linear-attn-decode-backend: flashinfer + enable-symm-mem: true + enable-cache-report: true + enable-metrics: true + context-length: 262144 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + page-size: 64 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + disaggregation-mode: decode + disaggregation-transfer-backend: mooncake + watchdog-timeout: 3600 + mamba-radix-cache-strategy: no_buffer + disable-radix-cache: true + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + scheduler-recv-interval: 10 + stream-interval: 50 + disaggregation-decode-extra-slots: 0 + disaggregation-decode-retraction-backup: cpu_tensor + sbatch_directives: + mem: '0' + cpus-per-task: '144' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + ENABLE_AGENTX_POWER: '0' + IS_MULTINODE: 'true' + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' + +# Colocated TP2+EP2 prefill and decode with write-through HiCache. +override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c32_mtp_hicache: + name: qwen35-b300-disagg-1p1d-tp2ep2-tp2ep2-colocated-c32-mtp-hicache + roles: + prefill: + gpus: 2 + args: + tensor-parallel-size: 2 + expert-parallel-size: 2 + max-running-requests: 64 + hicache-size: 298 + hicache-write-policy: write_through + decode: + gpus: 2 + args: + tensor-parallel-size: 2 + expert-parallel-size: 2 + mem-fraction-static: 0.88 + max-running-requests: 64 + cuda-graph-max-bs-decode: 32 + benchmark: + concurrencies: [32] + +# Colocated TP2+EP2 prefill and decode with write-through HiCache. +override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c40_mtp_hicache: + name: qwen35-b300-disagg-1p1d-tp2ep2-tp2ep2-colocated-c40-mtp-hicache + roles: + prefill: + gpus: 2 + args: + tensor-parallel-size: 2 + expert-parallel-size: 2 + max-running-requests: 64 + hicache-size: 298 + hicache-write-policy: write_through + decode: + gpus: 2 + args: + tensor-parallel-size: 2 + expert-parallel-size: 2 + mem-fraction-static: 0.88 + max-running-requests: 64 + cuda-graph-max-bs-decode: 48 + benchmark: + concurrencies: [40] + +# Colocated TP2+EP2 prefill and decode with write-through HiCache. +override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c44_mtp_hicache: + name: qwen35-b300-disagg-1p1d-tp2ep2-tp2ep2-colocated-c44-mtp-hicache + roles: + prefill: + gpus: 2 + args: + tensor-parallel-size: 2 + expert-parallel-size: 2 + max-running-requests: 64 + hicache-size: 298 + hicache-write-policy: write_through + decode: + gpus: 2 + args: + tensor-parallel-size: 2 + expert-parallel-size: 2 + mem-fraction-static: 0.88 + max-running-requests: 64 + cuda-graph-max-bs-decode: 48 + benchmark: + concurrencies: [44] + +# Colocated TP2+EP2 prefill and decode with write-through HiCache. +override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c48_mtp_hicache: + name: qwen35-b300-disagg-1p1d-tp2ep2-tp2ep2-colocated-c48-mtp-hicache + roles: + prefill: + gpus: 2 + args: + tensor-parallel-size: 2 + expert-parallel-size: 2 + max-running-requests: 64 + hicache-size: 298 + hicache-write-policy: write_through + decode: + gpus: 2 + args: + tensor-parallel-size: 2 + expert-parallel-size: 2 + mem-fraction-static: 0.92 + max-running-requests: 64 + cuda-graph-max-bs-decode: 48 + benchmark: + concurrencies: [48] + +# Colocated TP2+EP2 prefill and decode with write-through HiCache. +override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c56_replayssm_mtp_hicache: + name: qwen35-b300-disagg-1p1d-tp2ep2-tp2ep2-colocated-c56-replayssm-mtp-hicache + roles: + prefill: + gpus: 2 + args: + tensor-parallel-size: 2 + expert-parallel-size: 2 + max-running-requests: 64 + hicache-size: 298 + hicache-write-policy: write_through + decode: + gpus: 2 + args: + tensor-parallel-size: 2 + expert-parallel-size: 2 + mem-fraction-static: 0.92 + max-running-requests: 64 + cuda-graph-max-bs-decode: 48 + enable-linear-replayssm-spec: true + benchmark: + concurrencies: [56] + +# Colocated prefill and decode with prefill HiCache. +override_disagg_1p1d_tp4_tp4_colocated_c12_mtp_hicache: + name: qwen35-b300-disagg-1p1d-tp4-tp4-colocated-c12-mtp-hicache + roles: + prefill: + gpus: 4 + args: + tensor-parallel-size: 4 + expert-parallel-size: 1 + max-running-requests: 32 + hicache-size: 72 + hicache-write-policy: write_through_selective + decode: + gpus: 4 + args: + tensor-parallel-size: 4 + expert-parallel-size: 1 + mem-fraction-static: 0.8 + max-running-requests: 32 + cuda-graph-max-bs-decode: 16 + benchmark: + concurrencies: [12] + +# Colocated TP4 prefill and decode with prefill HiCache. +override_disagg_1p1d_tp4_tp4_colocated_c24_mtp_hicache: + name: qwen35-b300-disagg-1p1d-tp4-tp4-colocated-c24-mtp-hicache + roles: + prefill: + gpus: 4 + args: + tensor-parallel-size: 4 + expert-parallel-size: 1 + max-running-requests: 32 + hicache-size: 72 + hicache-write-policy: write_through_selective + decode: + gpus: 4 + args: + tensor-parallel-size: 4 + expert-parallel-size: 1 + mem-fraction-static: 0.8 + max-running-requests: 32 + cuda-graph-max-bs-decode: 32 + benchmark: + concurrencies: [24] + +# Colocated prefill and decode with prefill HiCache. +override_disagg_1p1d_tp4_tp4_colocated_c4_mtp_hicache: + name: qwen35-b300-disagg-1p1d-tp4-tp4-colocated-c4-mtp-hicache + roles: + prefill: + gpus: 4 + args: + tensor-parallel-size: 4 + expert-parallel-size: 1 + max-running-requests: 32 + hicache-size: 72 + hicache-write-policy: write_through_selective + decode: + gpus: 4 + args: + tensor-parallel-size: 4 + expert-parallel-size: 1 + mem-fraction-static: 0.8 + max-running-requests: 32 + cuda-graph-max-bs-decode: 16 + benchmark: + concurrencies: [4] + +# Colocated prefill and decode with prefill HiCache. +override_disagg_1p1d_tp4ep4_tp4_colocated_c32_mtp_hicache: + name: qwen35-b300-disagg-1p1d-tp4ep4-tp4-colocated-c32-mtp-hicache + roles: + prefill: + gpus: 4 + args: + tensor-parallel-size: 4 + expert-parallel-size: 4 + max-running-requests: 32 + hicache-size: 72 + hicache-write-policy: write_through_selective + decode: + gpus: 4 + args: + tensor-parallel-size: 4 + expert-parallel-size: 1 + mem-fraction-static: 0.8 + max-running-requests: 32 + cuda-graph-max-bs-decode: 32 + benchmark: + concurrencies: [32] diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache-cap48.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache-cap48.yaml deleted file mode 100644 index 6912536759..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache-cap48.yaml +++ /dev/null @@ -1,115 +0,0 @@ -schema: 2 -name: qwen35-gb200-dynsg-agentic-mtp-agg-tp2ep2-hicache-cap48 - -model: {path: qwen3.5-fp4, container: lmsysorg/sglang:nightly-dev-20260818-c0b6474b, precision: fp4} -identity: - model: {repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2} - container: {image: lmsysorg/sglang:nightly-dev-20260818-c0b6474b} -slurm: {time_limit: "8:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} -resources: {gpu_type: gb200, gpus_per_node: 4} -services: - - name: nats - type: nats - options: - max_payload_mb: 8 -dynamo: - install: true - source: - wheel: 1.4.2 - request_plane: tcp -environment: - PIP_BREAK_SYSTEM_PACKAGES: "1" -frontend: - type: dynamo - enable_multiple_frontends: false - env: {} - args: - router-mode: round-robin - router-session-affinity-ttl-secs: 3600 - -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 2 - env: - PYTHONNOUSERSITE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - SGLANG_ENABLE_SPEC_V2: "1" - SGL_ENABLE_JIT_DEEPGEMM: "false" - TORCH_CUDA_ARCH_LIST: "10.0" - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 2 - pipeline-parallel-size: 1 - data-parallel-size: 1 - expert-parallel-size: 2 - enable-dp-attention: false - enable-dp-lm-head: false - enable-symm-mem: false - quantization: modelopt_fp4 - fp4-gemm-backend: flashinfer_cutlass - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-radix-cache-strategy: extra_buffer_lazy - mamba-track-interval: 1048576 - attention-backend: trtllm_mha - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - moe-runner-backend: flashinfer_trtllm - speculative-moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - enable-linear-replayssm-spec: true - linear-replayssm-cache-len: 8 - cuda-graph-max-bs: 48 - max-running-requests: 48 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.88 - max-mamba-cache-size: 192 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - page-size: 64 - enable-hierarchical-cache: true - hicache-ratio: 2.5 - hicache-io-backend: direct - hicache-mem-layout: page_first_direct - hicache-write-policy: write_through_selective - mamba-max-states-per-path: 1 - weight-loader-prefetch-checkpoints: true - weight-loader-prefetch-num-threads: 4 - weight-loader-drop-cache-after-load: false - model-loader-extra-config: '{"enable_multithread_load":true}' - watchdog-timeout: 1000000 - enable-metrics: true - enable-cache-report: true - -sbatch_directives: {mem: "0", cpus-per-task: "144"} -srun_options: {mem: "0", container-remap-root: ""} -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "false" - TP: "2" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_EXTRA_INPUTS: '{"chat_template_kwargs":{"enable_thinking":true},"presence_penalty":0,"temperature":0.6,"top_k":20,"top_p":0.95}' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache-k3-baseline.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache-k3-baseline.yaml deleted file mode 100644 index 624e4e7f68..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache-k3-baseline.yaml +++ /dev/null @@ -1,116 +0,0 @@ -schema: 2 -name: qwen35-gb200-dynsg-agentic-mtp-agg-tp2ep2-hicache-k3-baseline - -model: {path: qwen3.5-fp4, container: lmsysorg/sglang:nightly-dev-20260818-c0b6474b, precision: fp4} -identity: - model: {repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2} - container: {image: lmsysorg/sglang:nightly-dev-20260818-c0b6474b} -slurm: {time_limit: "8:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} -resources: {gpu_type: gb200, gpus_per_node: 4} -services: - - name: nats - type: nats - options: - max_payload_mb: 8 -dynamo: - install: true - source: - wheel: 1.4.2 - request_plane: tcp -environment: - PIP_BREAK_SYSTEM_PACKAGES: "1" -frontend: - type: dynamo - enable_multiple_frontends: false - env: {} - args: - router-mode: round-robin - router-session-affinity-ttl-secs: 3600 - -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 2 - env: - PYTHONNOUSERSITE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - SGLANG_ENABLE_SPEC_V2: "1" - SGL_ENABLE_JIT_DEEPGEMM: "false" - TORCH_CUDA_ARCH_LIST: "10.0" - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 2 - pipeline-parallel-size: 1 - data-parallel-size: 1 - expert-parallel-size: 2 - enable-dp-attention: false - enable-dp-lm-head: false - enable-symm-mem: false - quantization: modelopt_fp4 - fp4-gemm-backend: flashinfer_cutlass - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-radix-cache-strategy: extra_buffer_lazy - mamba-track-interval: 1048576 - attention-backend: trtllm_mha - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - moe-runner-backend: flashinfer_trtllm - speculative-moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - enable-linear-replayssm-spec: true - linear-replayssm-cache-len: 8 - cuda-graph-max-bs: 64 - max-running-requests: 80 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.88 - max-mamba-cache-size: 320 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - tokenizer-worker-num: 1 - page-size: 64 - enable-hierarchical-cache: true - hicache-ratio: 2.5 - hicache-io-backend: direct - hicache-mem-layout: page_first_direct - hicache-write-policy: write_through_selective - mamba-max-states-per-path: 1 - weight-loader-prefetch-checkpoints: true - weight-loader-prefetch-num-threads: 4 - weight-loader-drop-cache-after-load: false - model-loader-extra-config: '{"enable_multithread_load":true}' - watchdog-timeout: 1000000 - enable-metrics: true - enable-cache-report: true - -sbatch_directives: {mem: "0", cpus-per-task: "144"} -srun_options: {mem: "0", container-remap-root: ""} -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "false" - TP: "2" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_EXTRA_INPUTS: '{"chat_template_kwargs":{"enable_thinking":true},"presence_penalty":0,"temperature":0.6,"top_k":20,"top_p":0.95}' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache.yaml deleted file mode 100644 index 35991d1a61..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache.yaml +++ /dev/null @@ -1,117 +0,0 @@ -schema: 2 -name: qwen35-gb200-dynsg-agentic-mtp-agg-tp2ep2-hicache-k5 - -model: {path: qwen3.5-fp4, container: lmsysorg/sglang:nightly-dev-20260818-c0b6474b, precision: fp4} -identity: - model: {repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2} - container: {image: lmsysorg/sglang:nightly-dev-20260818-c0b6474b} -slurm: {time_limit: "8:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} -resources: {gpu_type: gb200, gpus_per_node: 4} -services: - - name: nats - type: nats - options: - max_payload_mb: 8 -dynamo: - install: true - source: - wheel: 1.4.2 - request_plane: tcp -environment: - PIP_BREAK_SYSTEM_PACKAGES: "1" -frontend: - type: dynamo - enable_multiple_frontends: false - env: {} - args: - router-mode: round-robin - router-session-affinity-ttl-secs: 3600 - -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 2 - env: - PYTHONNOUSERSITE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - SGLANG_ENABLE_SPEC_V2: "1" - SGL_ENABLE_JIT_DEEPGEMM: "false" - TORCH_CUDA_ARCH_LIST: "10.0" - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 2 - pipeline-parallel-size: 1 - data-parallel-size: 1 - expert-parallel-size: 2 - enable-dp-attention: false - enable-dp-lm-head: false - enable-symm-mem: false - quantization: modelopt_fp4 - fp4-gemm-backend: flashinfer_cutlass - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-radix-cache-strategy: extra_buffer_lazy - mamba-track-interval: 1048576 - attention-backend: trtllm_mha - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - moe-runner-backend: flashinfer_trtllm - speculative-moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 5 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 6 - enable-linear-replayssm-spec: true - linear-replayssm-cache-len: 8 - cuda-graph-max-bs: 64 - max-running-requests: 80 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.88 - # extra_buffer_lazy uses four physical Mamba state slots per running request. - # This layout is the measured K5 middle frontier through C28. - max-mamba-cache-size: 320 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - page-size: 64 - enable-hierarchical-cache: true - hicache-ratio: 2.5 - hicache-io-backend: direct - hicache-mem-layout: page_first_direct - hicache-write-policy: write_through_selective - mamba-max-states-per-path: 1 - weight-loader-prefetch-checkpoints: true - weight-loader-prefetch-num-threads: 4 - weight-loader-drop-cache-after-load: false - model-loader-extra-config: '{"enable_multithread_load":true}' - watchdog-timeout: 1000000 - enable-metrics: true - enable-cache-report: true - -sbatch_directives: {mem: "0", cpus-per-task: "144"} -srun_options: {mem: "0", container-remap-root: ""} -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "false" - TP: "2" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_EXTRA_INPUTS: '{"chat_template_kwargs":{"enable_thinking":true},"presence_penalty":0,"temperature":0.6,"top_k":20,"top_p":0.95}' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-no-symm.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-no-symm.yaml deleted file mode 100644 index db3e9116f2..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-no-symm.yaml +++ /dev/null @@ -1,111 +0,0 @@ -schema: 2 -name: qwen35-gb200-dynsg-agentic-mtp-agg-tp4-k5-no-symm - -model: - path: qwen3.5-fp4 - container: lmsysorg/sglang:nightly-dev-20260818-c0b6474b - precision: fp4 - -identity: - model: {repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2} - container: {image: lmsysorg/sglang:nightly-dev-20260818-c0b6474b} - -slurm: {time_limit: "8:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} - -resources: - gpu_type: gb200 - gpus_per_node: 4 -services: - - name: nats - type: nats - options: - max_payload_mb: 8 -dynamo: - install: true - source: - wheel: 1.4.2 - request_plane: tcp -environment: - PIP_BREAK_SYSTEM_PACKAGES: "1" -frontend: - type: dynamo - enable_multiple_frontends: false - env: {} - args: - router-mode: round-robin - router-session-affinity-ttl-secs: 3600 - -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - - env: - PYTHONNOUSERSITE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - SGLANG_ENABLE_SPEC_V2: "1" - SGL_ENABLE_JIT_DEEPGEMM: "false" - TORCH_CUDA_ARCH_LIST: "10.0" - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-symm-mem: false - quantization: modelopt_fp4 - fp4-gemm-backend: flashinfer_cutlass - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 8192 - attention-backend: trtllm_mha - linear-attn-decode-backend: flashinfer - moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 5 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 6 - cuda-graph-max-bs: 64 - max-running-requests: 80 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.80 - max-mamba-cache-size: 360 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - mamba-max-states-per-path: 1 - weight-loader-prefetch-checkpoints: true - weight-loader-prefetch-num-threads: 4 - weight-loader-drop-cache-after-load: false - model-loader-extra-config: '{"enable_multithread_load":true}' - enable-metrics: true - enable-cache-report: true - -sbatch_directives: {mem: "0", cpus-per-task: "144"} -srun_options: {mem: "0", container-remap-root: ""} - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "false" - TP: "4" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_EXTRA_INPUTS: '{"chat_template_kwargs":{"enable_thinking":true},"presence_penalty":0,"temperature":0.6,"top_k":20,"top_p":0.95}' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-parity.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-parity.yaml deleted file mode 100644 index 4697c70678..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-parity.yaml +++ /dev/null @@ -1,114 +0,0 @@ -schema: 2 -name: qwen35-gb200-dynsg-agentic-mtp-agg-tp4-k5-parity - -model: - path: qwen3.5-fp4 - container: lmsysorg/sglang:nightly-dev-20260818-c0b6474b - precision: fp4 - -identity: - model: {repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2} - container: {image: lmsysorg/sglang:nightly-dev-20260818-c0b6474b} - -slurm: {time_limit: "8:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} - -resources: - gpu_type: gb200 - gpus_per_node: 4 -services: - - name: nats - type: nats - options: - max_payload_mb: 8 -dynamo: - install: true - source: - wheel: 1.4.2 - request_plane: tcp -environment: - PIP_BREAK_SYSTEM_PACKAGES: "1" -frontend: - type: dynamo - enable_multiple_frontends: false - env: {} - args: - router-mode: round-robin - router-session-affinity-ttl-secs: 3600 - -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - - env: - PYTHONNOUSERSITE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - SGLANG_ENABLE_SPEC_V2: "1" - SGL_ENABLE_JIT_DEEPGEMM: "false" - TORCH_CUDA_ARCH_LIST: "10.0" - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-symm-mem: true - quantization: modelopt_fp4 - fp4-gemm-backend: flashinfer_cutlass - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-scheduler-strategy: extra_buffer_lazy - mamba-track-interval: 1048576 - attention-backend: trtllm_mha - linear-attn-prefill-backend: flashinfer - linear-attn-decode-backend: flashinfer - moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 5 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 6 - enable-linear-replayssm-spec: true - linear-replayssm-cache-len: 8 - cuda-graph-max-bs: 64 - max-running-requests: 80 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.80 - max-mamba-cache-size: 360 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - mamba-max-states-per-path: 1 - weight-loader-prefetch-checkpoints: true - weight-loader-prefetch-num-threads: 4 - weight-loader-drop-cache-after-load: false - model-loader-extra-config: '{"enable_multithread_load":true}' - enable-metrics: true - enable-cache-report: true - -sbatch_directives: {mem: "0", cpus-per-task: "144"} -srun_options: {mem: "0", container-remap-root: ""} - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "false" - TP: "4" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_EXTRA_INPUTS: '{"chat_template_kwargs":{"enable_thinking":true},"presence_penalty":0,"temperature":0.6,"top_k":20,"top_p":0.95}' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp.yaml deleted file mode 100644 index ea6ea024e3..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp.yaml +++ /dev/null @@ -1,111 +0,0 @@ -schema: 2 -name: qwen35-gb200-dynsg-agentic-mtp-agg-tp4-k5 - -model: - path: qwen3.5-fp4 - container: lmsysorg/sglang:nightly-dev-20260818-c0b6474b - precision: fp4 - -identity: - model: {repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2} - container: {image: lmsysorg/sglang:nightly-dev-20260818-c0b6474b} - -slurm: {time_limit: "8:00:00"} -health_check: {max_attempts: 2160, interval_seconds: 10} - -resources: - gpu_type: gb200 - gpus_per_node: 4 -services: - - name: nats - type: nats - options: - max_payload_mb: 8 -dynamo: - install: true - source: - wheel: 1.4.2 - request_plane: tcp -environment: - PIP_BREAK_SYSTEM_PACKAGES: "1" -frontend: - type: dynamo - enable_multiple_frontends: false - env: {} - args: - router-mode: round-robin - router-session-affinity-ttl-secs: 3600 - -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - - env: - PYTHONNOUSERSITE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - SGLANG_ENABLE_SPEC_V2: "1" - SGL_ENABLE_JIT_DEEPGEMM: "false" - TORCH_CUDA_ARCH_LIST: "10.0" - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-symm-mem: true - quantization: modelopt_fp4 - fp4-gemm-backend: flashinfer_cutlass - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 8192 - attention-backend: trtllm_mha - linear-attn-decode-backend: flashinfer - moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 5 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 6 - cuda-graph-max-bs: 64 - max-running-requests: 80 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.80 - max-mamba-cache-size: 360 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - mamba-max-states-per-path: 1 - weight-loader-prefetch-checkpoints: true - weight-loader-prefetch-num-threads: 4 - weight-loader-drop-cache-after-load: false - model-loader-extra-config: '{"enable_multithread_load":true}' - enable-metrics: true - enable-cache-report: true - -sbatch_directives: {mem: "0", cpus-per-task: "144"} -srun_options: {mem: "0", container-remap-root: ""} - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "false" - TP: "4" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_EXTRA_INPUTS: '{"chat_template_kwargs":{"enable_thinking":true},"presence_penalty":0,"temperature":0.6,"top_k":20,"top_p":0.95}' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..3a87586947 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml @@ -0,0 +1,278 @@ +# srt-slurm recipes for qwen3.5/sglang/gb200-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: qwen3.5-fp4 + container: lmsysorg/sglang:nightly-dev-20260818-c0b6474b + precision: fp4 + identity: + model: + repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + container: + image: lmsysorg/sglang:nightly-dev-20260818-c0b6474b + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 2160 + interval_seconds: 10 + resources: + gpu_type: gb200 + gpus_per_node: 4 + services: + - name: nats + type: nats + options: + max_payload_mb: 8 + dynamo: + install: true + source: + wheel: 1.4.2 + request_plane: tcp + environment: + PIP_BREAK_SYSTEM_PACKAGES: '1' + frontend: + type: dynamo + enable_multiple_frontends: false + env: {} + args: + router-mode: round-robin + router-session-affinity-ttl-secs: 3600 + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + env: + PYTHONNOUSERSITE: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + TORCH_CUDA_ARCH_LIST: '10.0' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + model-path: /model/ + trust-remote-code: true + data-parallel-size: 1 + quantization: modelopt_fp4 + fp4-gemm-backend: flashinfer_cutlass + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + linear-attn-decode-backend: flashinfer + moe-runner-backend: flashinfer_trtllm + speculative-algorithm: NEXTN + speculative-eagle-topk: 1 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + allow-auto-truncate: true + stream-interval: 50 + scheduler-recv-interval: 10 + mamba-max-states-per-path: 1 + weight-loader-prefetch-checkpoints: true + weight-loader-prefetch-num-threads: 4 + weight-loader-drop-cache-after-load: false + model-loader-extra-config: '{"enable_multithread_load":true}' + enable-metrics: true + enable-cache-report: true + sbatch_directives: + mem: '0' + cpus-per-task: '144' + srun_options: + mem: '0' + container-remap-root: '' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'false' + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' + AIPERF_EXTRA_INPUTS: '{"chat_template_kwargs":{"enable_thinking":true},"presence_penalty":0,"temperature":0.6,"top_k":20,"top_p":0.95}' + +override_agg_tp2ep2_mtp_hicache_cap48: + name: qwen35-gb200-dynsg-agentic-mtp-agg-tp2ep2-hicache-cap48 + roles: + agg: + gpus: 2 + args: + tensor-parallel-size: 2 + pipeline-parallel-size: 1 + expert-parallel-size: 2 + enable-dp-attention: false + enable-dp-lm-head: false + enable-symm-mem: false + mamba-radix-cache-strategy: extra_buffer_lazy + mamba-track-interval: 1048576 + linear-attn-prefill-backend: flashinfer + speculative-moe-runner-backend: flashinfer_trtllm + speculative-num-steps: 3 + speculative-num-draft-tokens: 4 + enable-linear-replayssm-spec: true + linear-replayssm-cache-len: 8 + cuda-graph-max-bs: 48 + max-running-requests: 48 + mem-fraction-static: 0.88 + max-mamba-cache-size: 192 + page-size: 64 + enable-hierarchical-cache: true + hicache-ratio: 2.5 + hicache-io-backend: direct + hicache-mem-layout: page_first_direct + hicache-write-policy: write_through_selective + watchdog-timeout: 1000000 + benchmark: + env: + TP: '2' + +override_agg_tp2ep2_mtp_hicache_k3_baseline: + name: qwen35-gb200-dynsg-agentic-mtp-agg-tp2ep2-hicache-k3-baseline + roles: + agg: + gpus: 2 + args: + tensor-parallel-size: 2 + pipeline-parallel-size: 1 + expert-parallel-size: 2 + enable-dp-attention: false + enable-dp-lm-head: false + enable-symm-mem: false + mamba-radix-cache-strategy: extra_buffer_lazy + mamba-track-interval: 1048576 + linear-attn-prefill-backend: flashinfer + speculative-moe-runner-backend: flashinfer_trtllm + speculative-num-steps: 3 + speculative-num-draft-tokens: 4 + enable-linear-replayssm-spec: true + linear-replayssm-cache-len: 8 + cuda-graph-max-bs: 64 + max-running-requests: 80 + mem-fraction-static: 0.88 + max-mamba-cache-size: 320 + page-size: 64 + enable-hierarchical-cache: true + hicache-ratio: 2.5 + hicache-io-backend: direct + hicache-mem-layout: page_first_direct + hicache-write-policy: write_through_selective + watchdog-timeout: 1000000 + tokenizer-worker-num: 1 + benchmark: + env: + TP: '2' + +override_agg_tp2ep2_mtp_hicache: + name: qwen35-gb200-dynsg-agentic-mtp-agg-tp2ep2-hicache-k5 + roles: + agg: + gpus: 2 + args: + tensor-parallel-size: 2 + pipeline-parallel-size: 1 + expert-parallel-size: 2 + enable-dp-attention: false + enable-dp-lm-head: false + enable-symm-mem: false + mamba-radix-cache-strategy: extra_buffer_lazy + mamba-track-interval: 1048576 + linear-attn-prefill-backend: flashinfer + speculative-moe-runner-backend: flashinfer_trtllm + speculative-num-steps: 5 + speculative-num-draft-tokens: 6 + enable-linear-replayssm-spec: true + linear-replayssm-cache-len: 8 + cuda-graph-max-bs: 64 + max-running-requests: 80 + mem-fraction-static: 0.88 + # extra_buffer_lazy uses four physical Mamba state slots per running request. + # This layout is the measured K5 middle frontier through C28. + max-mamba-cache-size: 320 + page-size: 64 + enable-hierarchical-cache: true + hicache-ratio: 2.5 + hicache-io-backend: direct + hicache-mem-layout: page_first_direct + hicache-write-policy: write_through_selective + watchdog-timeout: 1000000 + benchmark: + env: + TP: '2' + +override_agg_tp4_mtp_no_symm: + name: qwen35-gb200-dynsg-agentic-mtp-agg-tp4-k5-no-symm + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + expert-parallel-size: 1 + enable-symm-mem: false + mamba-track-interval: 8192 + speculative-num-steps: 5 + speculative-num-draft-tokens: 6 + cuda-graph-max-bs: 64 + max-running-requests: 80 + mem-fraction-static: 0.8 + max-mamba-cache-size: 360 + mamba-scheduler-strategy: extra_buffer + benchmark: + env: + TP: '4' + +override_agg_tp4_mtp_parity: + name: qwen35-gb200-dynsg-agentic-mtp-agg-tp4-k5-parity + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + expert-parallel-size: 1 + enable-symm-mem: true + mamba-track-interval: 1048576 + linear-attn-prefill-backend: flashinfer + speculative-num-steps: 5 + speculative-num-draft-tokens: 6 + enable-linear-replayssm-spec: true + linear-replayssm-cache-len: 8 + cuda-graph-max-bs: 64 + max-running-requests: 80 + mem-fraction-static: 0.8 + max-mamba-cache-size: 360 + mamba-scheduler-strategy: extra_buffer_lazy + benchmark: + env: + TP: '4' + +override_agg_tp4_mtp: + name: qwen35-gb200-dynsg-agentic-mtp-agg-tp4-k5 + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + expert-parallel-size: 1 + enable-symm-mem: true + mamba-track-interval: 8192 + speculative-num-steps: 5 + speculative-num-draft-tokens: 6 + cuda-graph-max-bs: 64 + max-running-requests: 80 + mem-fraction-static: 0.8 + max-mamba-cache-size: 360 + mamba-scheduler-strategy: extra_buffer + benchmark: + env: + TP: '4' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c1x2x8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c1x2x8-mtp.yaml deleted file mode 100644 index 8dd9e66955..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c1x2x8-mtp.yaml +++ /dev/null @@ -1,156 +0,0 @@ -# Qwen3.5 FP8 GB200 disaggregated MTP 1P1D TP4/TP4 topology. - -schema: 2 -name: "qwen3.5-fp8-gb200-mtp-8k1k-1p1d-tp4-tp4" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - wheel: "1.5.0.dev20260917" -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 1 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - attention-backend: "trtllm_mha" - tensor-parallel-size: 4 - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - moe-runner-backend: "flashinfer_trtllm" - disable-radix-cache: true - max-running-requests: 1024 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - context-length: 16384 - cuda-graph-max-bs-decode: 1024 - decode-log-interval: 1 - stream-interval: 50 - disaggregation-mode: "prefill" - - decode: - nodes: 1 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - attention-backend: "trtllm_mha" - tensor-parallel-size: 4 - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - max-mamba-cache-size: 256 - moe-runner-backend: "flashinfer_trtllm" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disable-radix-cache: true - max-running-requests: 128 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - context-length: 16384 - cuda-graph-max-bs-decode: 1024 - decode-log-interval: 1 - stream-interval: 50 - disaggregation-mode: "decode" - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - random_range_ratio: 0.8 - concurrencies: "1x2x8" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml deleted file mode 100644 index 34b6f53606..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Qwen3.5 FP8 GB200 disaggregated MTP 1P1D TEP8/TEP8 points. - -schema: 2 -name: "qwen3.5-fp8-gb200-mtp-8k1k-1p1d-tep8-tep8" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - wheel: "1.5.0.dev20260917" -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 1 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 2 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - attention-backend: "trtllm_mha" - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 8 - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - moe-runner-backend: "flashinfer_trtllm" - disable-radix-cache: true - max-running-requests: 1024 - mem-fraction-static: 0.8 - max-total-tokens: 128000 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - context-length: 9236 - cuda-graph-max-bs-decode: 320 - scheduler-recv-interval: 10 - decode-log-interval: 50 - stream-interval: 50 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: "mooncake" - - decode: - nodes: 2 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - attention-backend: "trtllm_mha" - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 8 - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - max-mamba-cache-size: 1024 - moe-runner-backend: "flashinfer_trtllm" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disable-radix-cache: true - max-running-requests: 1024 - mem-fraction-static: 0.8 - max-total-tokens: 2200000 - chunked-prefill-size: 4096 - max-prefill-tokens: 16384 - context-length: 9236 - cuda-graph-max-bs-decode: 320 - scheduler-recv-interval: 10 - decode-log-interval: 50 - stream-interval: 50 - disaggregation-mode: "decode" - disaggregation-transfer-backend: "mooncake" - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "32x48x80" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-tp4-tp4-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-tp4-tp4-stp.yaml deleted file mode 100644 index e236b684a4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-tp4-tp4-stp.yaml +++ /dev/null @@ -1,134 +0,0 @@ -schema: 2 -name: "qwen3.5-1p1d-tp4-tp4" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - wheel: "1.5.0.dev20260917" -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 1 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - MC_FORCE_MNNVL: "1" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - attention-backend: "trtllm_mha" - tensor-parallel-size: 4 - mamba-ssm-dtype: "bfloat16" - moe-runner-backend: "flashinfer_trtllm" - disable-radix-cache: true - max-running-requests: 1024 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - context-length: 16384 - cuda-graph-max-bs-decode: 1024 - decode-log-interval: 1 - stream-interval: 50 - disaggregation-mode: "prefill" - - decode: - nodes: 1 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_HEALTH_CHECK_TIMEOUT: "3600" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - attention-backend: "trtllm_mha" - tensor-parallel-size: 4 - mamba-ssm-dtype: "bfloat16" - moe-runner-backend: "flashinfer_trtllm" - disable-radix-cache: true - max-running-requests: 1024 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - context-length: 16384 - cuda-graph-max-bs-decode: 1024 - decode-log-interval: 1 - stream-interval: 50 - disaggregation-mode: "decode" - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - random_range_ratio: 0.8 - concurrencies: "1x2x4x8x16x32x64x128" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml deleted file mode 100644 index 6ad51de7ce..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml +++ /dev/null @@ -1,185 +0,0 @@ -# Qwen3.5 FP8 GB200 disaggregated MTP 3P1D DEP4/DEP16 point. - -schema: 2 -name: "qwen3.5-fp8-gb200-mtp-8k1k-3p1d-dep4-dep16" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - wheel: "1.5.0.dev20260917" -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 2 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 3 - workers: 3 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - mem-fraction-static: 0.7 - chunked-prefill-size: 65536 - max-prefill-tokens: 16384 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - moe-dense-tp-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: "follow_bootstrap_room" - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.7 - max-mamba-cache-size: 1024 - max-running-requests: 1024 - cuda-graph-max-bs-decode: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "480" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-4p1d-dep4-dep16-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-4p1d-dep4-dep16-stp.yaml deleted file mode 100644 index b8c7f9aab0..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-4p1d-dep4-dep16-stp.yaml +++ /dev/null @@ -1,172 +0,0 @@ -schema: 2 -name: "qwen3.5-4p1d-dep4-dep16" - -setup_script: rebuild-deepep.sh - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - wheel: "1.5.0.dev20260917" -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 3 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 4 - workers: 4 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - mem-fraction-static: 0.8 - chunked-prefill-size: 65536 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: "follow_bootstrap_room" - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.80 - max-mamba-cache-size: 2048 - max-running-requests: 2048 - cuda-graph-max-bs-decode: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "1024" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml deleted file mode 100644 index df5dc71f9b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml +++ /dev/null @@ -1,185 +0,0 @@ -# Qwen3.5 FP8 GB200 disaggregated MTP 4P1D DEP4/DEP16 point. - -schema: 2 -name: "qwen3.5-fp8-gb200-mtp-8k1k-4p1d-dep4-dep16" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - wheel: "1.5.0.dev20260917" -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 3 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 4 - workers: 4 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - mem-fraction-static: 0.7 - chunked-prefill-size: 98304 - max-prefill-tokens: 24576 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - moe-dense-tp-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: "follow_bootstrap_room" - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.7 - max-mamba-cache-size: 1024 - max-running-requests: 1024 - cuda-graph-max-bs-decode: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "768" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml deleted file mode 100644 index 0b8d6ea9f9..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml +++ /dev/null @@ -1,185 +0,0 @@ -# Qwen3.5 FP8 GB200 disaggregated MTP 6P1D DEP4/DEP16 point. - -schema: 2 -name: "qwen3.5-fp8-gb200-mtp-8k1k-6p1d-dep4-dep16" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - wheel: "1.5.0.dev20260917" -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 5 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 6 - workers: 6 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31001 - mem-fraction-static: 0.7 - chunked-prefill-size: 65536 - max-prefill-tokens: 16384 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - moe-dense-tp-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: "follow_bootstrap_room" - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31001 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.7 - max-mamba-cache-size: 1024 - max-running-requests: 1024 - cuda-graph-max-bs-decode: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "1280" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml deleted file mode 100644 index c68b98a501..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml +++ /dev/null @@ -1,185 +0,0 @@ -# Qwen3.5 FP8 GB200 disaggregated MTP 7P1D DEP4/DEP16 point. - -schema: 2 -name: "qwen3.5-fp8-gb200-mtp-8k1k-7p1d-dep4-dep16" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - wheel: "1.5.0.dev20260917" -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 6 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 7 - workers: 7 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31001 - mem-fraction-static: 0.7 - chunked-prefill-size: 65536 - max-prefill-tokens: 16384 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - moe-dense-tp-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: "follow_bootstrap_room" - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31001 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.7 - max-mamba-cache-size: 2048 - max-running-requests: 2048 - cuda-graph-max-bs-decode: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "1344" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-8p1d-dep4-dep16-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-8p1d-dep4-dep16-stp.yaml deleted file mode 100644 index fcb014b91b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-8p1d-dep4-dep16-stp.yaml +++ /dev/null @@ -1,168 +0,0 @@ -schema: 2 -name: "qwen3.5-8p1d-dep4-dep16" - -setup_script: rebuild-deepep.sh - -sbatch_directives: - mem: "0" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated -dynamo: - install: true - - source: - wheel: "1.5.0.dev20260917" -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 4 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 8 - workers: 8 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - mem-fraction-static: 0.8 - chunked-prefill-size: 65536 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: "follow_bootstrap_room" - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.80 - max-mamba-cache-size: 2048 - max-running-requests: 2048 - cuda-graph-max-bs-decode: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - random_range_ratio: 0.8 - concurrencies: "2048x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml deleted file mode 100644 index 1642c1796f..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml +++ /dev/null @@ -1,185 +0,0 @@ -# Qwen3.5 FP8 GB200 disaggregated MTP 8P1D DEP4/DEP16 points. - -schema: 2 -name: "qwen3.5-fp8-gb200-mtp-8k1k-8p1d-dep4-dep16" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - wheel: "1.5.0.dev20260917" -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 7 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" - precision: "fp8" - -resources: - gpu_type: "gb200" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 8 - workers: 8 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31001 - mem-fraction-static: 0.7 - chunked-prefill-size: 98304 - max-prefill-tokens: 24576 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - moe-dense-tp-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: "follow_bootstrap_room" - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31001 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.7 - max-mamba-cache-size: 2048 - max-running-requests: 2048 - cuda-graph-max-bs-decode: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "1920x2304" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 9401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml new file mode 100644 index 0000000000..0533bbf3e8 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml @@ -0,0 +1,881 @@ +# srt-slurm recipes for qwen3.5/sglang/gb200-fp8/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_. + +schema: 2 + +base: + sbatch_directives: + mem: '0' + dynamo: + install: true + source: + wheel: 1.5.0.dev20260917 + frontend: + type: dynamo + enable_multiple_frontends: true + nginx_container: nginx + model: + path: qwen3.5-fp8 + container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 + precision: fp8 + resources: + gpu_type: gb200 + gpus_per_node: 4 + engine: sglang + roles: + prefill: + env: + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '3600' + TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: '3600' + TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: '3600' + PYTHONUNBUFFERED: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + MC_FORCE_MNNVL: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + args: + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + model-path: /model/ + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + trust-remote-code: true + attention-backend: trtllm_mha + mamba-ssm-dtype: bfloat16 + moe-runner-backend: flashinfer_trtllm + disable-radix-cache: true + disaggregation-mode: prefill + decode: + workers: 1 + env: + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '3600' + TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: '3600' + TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: '3600' + PYTHONUNBUFFERED: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + MC_FORCE_MNNVL: '1' + MC_TE_METRIC: 'true' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + model-path: /model/ + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + trust-remote-code: true + attention-backend: trtllm_mha + mamba-ssm-dtype: bfloat16 + disable-radix-cache: true + stream-interval: 50 + disaggregation-mode: decode + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + random_range_ratio: 0.8 + +# Qwen3.5 FP8 GB200 disaggregated MTP 1P1D TP4/TP4 topology. +override_disagg_1p1d_p_tp4_d_tp4_b128_c1x2x8_mtp: + name: qwen3.5-fp8-gb200-mtp-8k1k-1p1d-tp4-tp4 + frontend: + num_additional_frontends: 1 + roles: + prefill: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 2048 + max-running-requests: 1024 + mem-fraction-static: 0.8 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + context-length: 16384 + cuda-graph-max-bs-decode: 1024 + decode-log-interval: 1 + stream-interval: 50 + decode: + nodes: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + args: + tensor-parallel-size: 4 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 256 + moe-runner-backend: flashinfer_trtllm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 128 + mem-fraction-static: 0.8 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + context-length: 16384 + cuda-graph-max-bs-decode: 1024 + decode-log-interval: 1 + benchmark: + concurrencies: 1x2x8 + use_chat_template: true + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + +# Qwen3.5 FP8 GB200 disaggregated MTP 1P1D TEP8/TEP8 points. +override_disagg_1p1d_p_tp8_ep8_d_tp8_ep8_b1024_c32x48x80_mtp: + name: qwen3.5-fp8-gb200-mtp-8k1k-1p1d-tep8-tep8 + frontend: + num_additional_frontends: 1 + roles: + prefill: + nodes: 2 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 8 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 2048 + max-running-requests: 1024 + mem-fraction-static: 0.8 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + context-length: 9236 + cuda-graph-max-bs-decode: 320 + decode-log-interval: 50 + stream-interval: 50 + random-seed: 42 + data-parallel-size: 1 + expert-parallel-size: 8 + max-total-tokens: 128000 + scheduler-recv-interval: 10 + disaggregation-transfer-backend: mooncake + decode: + nodes: 2 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + args: + tensor-parallel-size: 8 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 1024 + moe-runner-backend: flashinfer_trtllm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 1024 + mem-fraction-static: 0.8 + chunked-prefill-size: 4096 + max-prefill-tokens: 16384 + context-length: 9236 + cuda-graph-max-bs-decode: 320 + decode-log-interval: 50 + random-seed: 42 + data-parallel-size: 1 + expert-parallel-size: 8 + max-total-tokens: 2200000 + scheduler-recv-interval: 10 + disaggregation-transfer-backend: mooncake + benchmark: + concurrencies: 32x48x80 + use_chat_template: true + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + +override_disagg_1p1d_tp4_tp4_stp: + name: qwen3.5-1p1d-tp4-tp4 + frontend: + num_additional_frontends: 1 + roles: + prefill: + nodes: 1 + workers: 1 + args: + tensor-parallel-size: 4 + max-running-requests: 1024 + mem-fraction-static: 0.8 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + context-length: 16384 + cuda-graph-max-bs-decode: 1024 + decode-log-interval: 1 + stream-interval: 50 + decode: + nodes: 1 + env: + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_HEALTH_CHECK_TIMEOUT: '3600' + args: + tensor-parallel-size: 4 + moe-runner-backend: flashinfer_trtllm + max-running-requests: 1024 + mem-fraction-static: 0.8 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + context-length: 16384 + cuda-graph-max-bs-decode: 1024 + decode-log-interval: 1 + benchmark: + concurrencies: 1x2x4x8x16x32x64x128 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + +# Qwen3.5 FP8 GB200 disaggregated MTP 3P1D DEP4/DEP16 point. +override_disagg_3p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c480_mtp: + name: qwen3.5-fp8-gb200-mtp-8k1k-3p1d-dep4-dep16 + frontend: + num_additional_frontends: 2 + roles: + prefill: + nodes: 3 + workers: 3 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 65536 + max-prefill-tokens: 16384 + random-seed: 42 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + decode: + nodes: 4 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 1024 + moe-runner-backend: deep_gemm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 1024 + mem-fraction-static: 0.7 + chunked-prefill-size: 4096 + context-length: 16384 + cuda-graph-max-bs-decode: 128 + decode-log-interval: 1 + random-seed: 42 + data-parallel-size: 16 + expert-parallel-size: 16 + moe-dense-tp-size: 1 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: follow_bootstrap_room + disaggregation-bootstrap-port: 31000 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + benchmark: + concurrencies: '480' + use_chat_template: true + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + +override_disagg_4p1d_dep4_dep16_stp: + name: qwen3.5-4p1d-dep4-dep16 + frontend: + num_additional_frontends: 3 + roles: + prefill: + nodes: 4 + workers: 4 + env: + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 2048 + mem-fraction-static: 0.8 + chunked-prefill-size: 65536 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + decode: + nodes: 4 + env: + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 2048 + moe-runner-backend: deep_gemm + max-running-requests: 2048 + mem-fraction-static: 0.8 + chunked-prefill-size: 4096 + context-length: 16384 + cuda-graph-max-bs-decode: 128 + decode-log-interval: 1 + data-parallel-size: 16 + expert-parallel-size: 16 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: follow_bootstrap_room + disaggregation-bootstrap-port: 31000 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + benchmark: + concurrencies: '1024' + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + setup_script: rebuild-deepep.sh + +# Qwen3.5 FP8 GB200 disaggregated MTP 4P1D DEP4/DEP16 point. +override_disagg_4p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c768_mtp: + name: qwen3.5-fp8-gb200-mtp-8k1k-4p1d-dep4-dep16 + frontend: + num_additional_frontends: 3 + roles: + prefill: + nodes: 4 + workers: 4 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 98304 + max-prefill-tokens: 24576 + random-seed: 42 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + decode: + nodes: 4 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 1024 + moe-runner-backend: deep_gemm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 1024 + mem-fraction-static: 0.7 + chunked-prefill-size: 4096 + context-length: 16384 + cuda-graph-max-bs-decode: 128 + decode-log-interval: 1 + random-seed: 42 + data-parallel-size: 16 + expert-parallel-size: 16 + moe-dense-tp-size: 1 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: follow_bootstrap_room + disaggregation-bootstrap-port: 31000 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + benchmark: + concurrencies: '768' + use_chat_template: true + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + +# Qwen3.5 FP8 GB200 disaggregated MTP 6P1D DEP4/DEP16 point. +override_disagg_6p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c1280_mtp: + name: qwen3.5-fp8-gb200-mtp-8k1k-6p1d-dep4-dep16 + frontend: + num_additional_frontends: 5 + roles: + prefill: + nodes: 6 + workers: 6 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 65536 + max-prefill-tokens: 16384 + random-seed: 42 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + disaggregation-bootstrap-port: 31001 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + decode: + nodes: 4 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 1024 + moe-runner-backend: deep_gemm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 1024 + mem-fraction-static: 0.7 + chunked-prefill-size: 4096 + context-length: 16384 + cuda-graph-max-bs-decode: 128 + decode-log-interval: 1 + random-seed: 42 + data-parallel-size: 16 + expert-parallel-size: 16 + moe-dense-tp-size: 1 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: follow_bootstrap_room + disaggregation-bootstrap-port: 31001 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + benchmark: + concurrencies: '1280' + use_chat_template: true + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + +# Qwen3.5 FP8 GB200 disaggregated MTP 7P1D DEP4/DEP16 point. +override_disagg_7p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b2048_c1344_mtp: + name: qwen3.5-fp8-gb200-mtp-8k1k-7p1d-dep4-dep16 + frontend: + num_additional_frontends: 6 + roles: + prefill: + nodes: 7 + workers: 7 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 65536 + max-prefill-tokens: 16384 + random-seed: 42 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + disaggregation-bootstrap-port: 31001 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + decode: + nodes: 4 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 2048 + moe-runner-backend: deep_gemm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 4096 + context-length: 16384 + cuda-graph-max-bs-decode: 128 + decode-log-interval: 1 + random-seed: 42 + data-parallel-size: 16 + expert-parallel-size: 16 + moe-dense-tp-size: 1 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: follow_bootstrap_room + disaggregation-bootstrap-port: 31001 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + benchmark: + concurrencies: '1344' + use_chat_template: true + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + +override_disagg_8p1d_dep4_dep16_stp: + name: qwen3.5-8p1d-dep4-dep16 + frontend: + num_additional_frontends: 4 + roles: + prefill: + nodes: 8 + workers: 8 + env: + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 2048 + mem-fraction-static: 0.8 + chunked-prefill-size: 65536 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + decode: + nodes: 4 + env: + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 2048 + moe-runner-backend: deep_gemm + max-running-requests: 2048 + mem-fraction-static: 0.8 + chunked-prefill-size: 4096 + context-length: 16384 + cuda-graph-max-bs-decode: 128 + decode-log-interval: 1 + data-parallel-size: 16 + expert-parallel-size: 16 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: follow_bootstrap_room + disaggregation-bootstrap-port: 31000 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + benchmark: + concurrencies: 2048x4096 + setup_script: rebuild-deepep.sh + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + +# Qwen3.5 FP8 GB200 disaggregated MTP 8P1D DEP4/DEP16 points. +override_disagg_8p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b2048_c1920x2304_mtp: + name: qwen3.5-fp8-gb200-mtp-8k1k-8p1d-dep4-dep16 + frontend: + num_additional_frontends: 7 + roles: + prefill: + nodes: 8 + workers: 8 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 98304 + max-prefill-tokens: 24576 + random-seed: 42 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + disaggregation-bootstrap-port: 31001 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + decode: + nodes: 4 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-radix-cache-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 2048 + moe-runner-backend: deep_gemm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 4096 + context-length: 16384 + cuda-graph-max-bs-decode: 128 + decode-log-interval: 1 + random-seed: 42 + data-parallel-size: 16 + expert-parallel-size: 16 + moe-dense-tp-size: 1 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: follow_bootstrap_room + disaggregation-bootstrap-port: 31001 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + benchmark: + concurrencies: 1920x2304 + use_chat_template: true + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c1x4x8x16x32x64x256-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c1x4x8x16x32x64x256-stp.yaml deleted file mode 100644 index 9933e469d4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c1x4x8x16x32x64x256-stp.yaml +++ /dev/null @@ -1,190 +0,0 @@ -# Qwen3.5-397B-A17B-NVFP4-V2 Disaggregated 1P1D: TP4 Prefill + TP4 Decode -# Pure tensor parallel, no expert parallel (STP) -# 8k1k sa-bench concurrency sweep on GB300 -# -# Values taken from ni_experiment_config of the -# sa-qwen-3.5-8k1k-fp4-baseline-low-latency study, row -# qwen3.5-1p_tp4x1d_tp4-aligned-ccsweep (CSV pareto export 2026-06-05). - -schema: 2 -name: "gb300-fp4-qwen3.5_8k1k_lowlat_0" - -model: - path: "qwen3.5-fp4" - container: "dynamo-sglang" - precision: "fp4" - -dynamo: - source: - pypi: "1.1.0" - -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 2 - nginx_container: nginx - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_ENABLE_SPEC_V2: "1" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_ENABLE_JIT_DEEPGEMM: "true" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - FLASHINFER_DISABLE_VERSION_CHECK: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - - args: - served-model-name: "nvidia/Qwen3.5-397B-A17B-NVFP4-V2" - model-path: "/model/" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - - reasoning-parser: "qwen3" - tool-call-parser: "qwen3_coder" - - quantization: "modelopt_fp4" - fp4-gemm-backend: "flashinfer_cutlass" - kv-cache-dtype: "fp8_e4m3" - - mamba-scheduler-strategy: "no_buffer" - mamba-ssm-dtype: "bfloat16" - mamba-track-interval: 2048 - - attention-backend: "trtllm_mha" - mm-attention-backend: "triton_attn" - moe-runner-backend: "flashinfer_trtllm" - linear-attn-decode-backend: "flashinfer" - - disaggregation-mode: "prefill" - disable-radix-cache: true - - mem-fraction-static: 0.8 - context-length: 9236 - max-total-tokens: 128000 - max-running-requests: 128 - cuda-graph-max-bs: 4 - chunked-prefill-size: 32768 - max-prefill-tokens: 32768 - scheduler-recv-interval: 10 - stream-interval: 30 - load-balance-method: "round_robin" - page-size: 64 - watchdog-timeout: 1000000 - log-level: "info" - - decode: - nodes: 1 - workers: 1 - - env: - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_ENABLE_SPEC_V2: "1" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_ENABLE_JIT_DEEPGEMM: "true" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - FLASHINFER_DISABLE_VERSION_CHECK: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "nvidia/Qwen3.5-397B-A17B-NVFP4-V2" - model-path: "/model/" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - - reasoning-parser: "qwen3" - tool-call-parser: "qwen3_coder" - - quantization: "modelopt_fp4" - fp4-gemm-backend: "flashinfer_cutlass" - kv-cache-dtype: "fp8_e4m3" - - mamba-scheduler-strategy: "no_buffer" - mamba-ssm-dtype: "bfloat16" - mamba-track-interval: 128 - - attention-backend: "trtllm_mha" - mm-attention-backend: "triton_attn" - moe-runner-backend: "flashinfer_trtllm" - linear-attn-decode-backend: "flashinfer" - - disaggregation-mode: "decode" - disable-radix-cache: true - - mem-fraction-static: 0.8 - context-length: 9236 - max-total-tokens: 1500000 - max-mamba-cache-size: 256 - max-running-requests: 128 - cuda-graph-max-bs: 256 - chunked-prefill-size: 32768 - max-prefill-tokens: 32768 - scheduler-recv-interval: 10 - stream-interval: 30 - page-size: 64 - watchdog-timeout: 1000000 - decode-log-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "1x4x8x16x32x64x256" - req_rate: "inf" - random_range_ratio: 0.8 - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - # 9401 is already bound by the cluster-level exporter on im-gb300 nodes; - # use a port outside that range. - port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-5p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b4096-c2048-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-5p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b4096-c2048-stp.yaml deleted file mode 100644 index 87f2142aee..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-5p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b4096-c2048-stp.yaml +++ /dev/null @@ -1,190 +0,0 @@ -# Qwen3.5-397B-A17B-NVFP4-V2 Disaggregated 5P1D wide-EP -# Prefill: 5 workers @ TP4/DP4/EP4 with DP-attn (per-node, DEP4) -# Decode: 1 worker @ TP16/DP16/EP16 with DP-attn + TBO (DEP16, 4 nodes) -# Total: 36 GB300 GPUs (5*4 + 4*4); 8k1k concurrency 1024/2048/3072. -# -# Values taken from ni_experiment_config of pareto row -# qwen3.5-dep16-fia2a-tbo-cc1024x2048x3072-dynamo-tot-nixl -# (sa-qwen-3.5-8k1k-fp4-baseline-mid-pareto study). - -schema: 2 -name: "gb300-fp4-qwen3.5_8k1k_maxtpt_0" - -model: - path: "qwen3.5-fp4" - container: "dynamo-sglang" - precision: "fp4" - -dynamo: - source: - pypi: "1.1.0" - -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 2 - nginx_container: nginx - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 5 - workers: 5 - env: - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_ENABLE_SPEC_V2: "1" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_ENABLE_JIT_DEEPGEMM: "true" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: "cutlass" - FLASHINFER_DISABLE_VERSION_CHECK: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_ENABLE_NIXL: "1" - - args: - served-model-name: "nvidia/Qwen3.5-397B-A17B-NVFP4-V2" - model-path: "/model/" - trust-remote-code: true - - quantization: "modelopt_fp4" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - disaggregation-transfer-backend: "nixl" - - mem-fraction-static: 0.8 - max-total-tokens: 128000 - chunked-prefill-size: 65536 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - log-level: "info" - page-size: 64 - - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - linear-attn-decode-backend: "flashinfer" - - decode: - nodes: 4 - workers: 1 - - env: - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_ENABLE_SPEC_V2: "1" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_ENABLE_JIT_DEEPGEMM: "true" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: "cutlass" - SGLANG_MOE_NVFP4_DISPATCH: "1" - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: "1" - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: "1" - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: "1" - FLASHINFER_DISABLE_VERSION_CHECK: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "1024" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - SGLANG_ENABLE_NIXL: "1" - - args: - served-model-name: "nvidia/Qwen3.5-397B-A17B-NVFP4-V2" - model-path: "/model/" - trust-remote-code: true - - quantization: "modelopt_fp4" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - enable-dp-attention: true - enable-dp-lm-head: true - enable-two-batch-overlap: true - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - disaggregation-transfer-backend: "nixl" - - chunked-prefill-size: 4096 - max-mamba-cache-size: 4096 - max-total-tokens: 2200000 - max-running-requests: 4096 - mem-fraction-static: 0.8 - watchdog-timeout: 1000000 - page-size: 64 - - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_cutedsl" - moe-a2a-backend: "flashinfer" - disable-shared-experts-fusion: true - linear-attn-decode-backend: "flashinfer" - - decode-log-interval: 50 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "2048" - req_rate: "inf" - random_range_ratio: 0.8 - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b5120-c5120-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b5120-c5120-stp.yaml deleted file mode 100644 index a6352aecb7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b5120-c5120-stp.yaml +++ /dev/null @@ -1,188 +0,0 @@ -# Qwen3.5-397B-A17B-NVFP4-V2 Disaggregated 6P1D wide-EP -# Prefill: 6 workers @ TP4/DP4/EP4 with DP-attn (per-node, DEP4) -# Decode: 1 worker @ TP16/DP16/EP16 with DP-attn + TBO (DEP16, 4 nodes) -# Total: 40 GB300 GPUs (6*4 + 4*4); 8k1k concurrency 5120. -# -# Values taken from ni_experiment_config of pareto row -# qwen3.5-6p_dep4x1d_dep16-fia2a-tbo-cc5120-dynamo-tot-mooncake -# (sa-qwen-3.5-8k1k-fp4-baseline-mid-pareto study). - -schema: 2 -name: "gb300-fp4-qwen3.5_8k1k_maxtpt_1" - -model: - path: "qwen3.5-fp4" - container: "dynamo-sglang" - precision: "fp4" - -dynamo: - source: - pypi: "1.1.0" - -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 2 - nginx_container: nginx - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 6 - workers: 6 - env: - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_ENABLE_SPEC_V2: "1" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_ENABLE_JIT_DEEPGEMM: "true" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: "cutlass" - FLASHINFER_DISABLE_VERSION_CHECK: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - - args: - served-model-name: "nvidia/Qwen3.5-397B-A17B-NVFP4-V2" - model-path: "/model/" - trust-remote-code: true - - quantization: "modelopt_fp4" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - disaggregation-transfer-backend: "mooncake" - - mem-fraction-static: 0.8 - max-total-tokens: 128000 - chunked-prefill-size: 65536 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - log-level: "info" - page-size: 64 - - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - linear-attn-decode-backend: "flashinfer" - - decode: - nodes: 4 - workers: 1 - - env: - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_ENABLE_SPEC_V2: "1" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_ENABLE_JIT_DEEPGEMM: "true" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: "cutlass" - SGLANG_MOE_NVFP4_DISPATCH: "1" - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: "1" - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: "1" - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: "1" - FLASHINFER_DISABLE_VERSION_CHECK: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "1024" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "nvidia/Qwen3.5-397B-A17B-NVFP4-V2" - model-path: "/model/" - trust-remote-code: true - - quantization: "modelopt_fp4" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - enable-dp-attention: true - enable-dp-lm-head: true - enable-two-batch-overlap: true - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - disaggregation-transfer-backend: "mooncake" - - chunked-prefill-size: 5120 - max-mamba-cache-size: 5120 - max-total-tokens: 3200000 - max-running-requests: 5120 - mem-fraction-static: 0.8 - watchdog-timeout: 1000000 - page-size: 64 - - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_cutedsl" - moe-a2a-backend: "flashinfer" - disable-shared-experts-fusion: true - linear-attn-decode-backend: "flashinfer" - - decode-log-interval: 50 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "5120" - req_rate: "inf" - random_range_ratio: 0.8 - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b5120-c5120-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b5120-c5120-stp.yaml deleted file mode 100644 index 942e00dd8d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b5120-c5120-stp.yaml +++ /dev/null @@ -1,188 +0,0 @@ -# Qwen3.5-397B-A17B-NVFP4-V2 Disaggregated 7P1D wide-EP -# Prefill: 7 workers @ TP4/DP4/EP4 with DP-attn (per-node, DEP4) -# Decode: 1 worker @ TP16/DP16/EP16 with DP-attn + TBO (DEP16, 4 nodes) -# Total: 44 GB300 GPUs (7*4 + 4*4); 8k1k concurrency 5120. -# -# Values taken from ni_experiment_config of pareto row -# qwen3.5-7p_dep4x1d_dep16-fia2a-tbo-cc5120-dynamo-tot-mooncake -# (sa-qwen-3.5-8k1k-fp4-baseline-mid-pareto study). - -schema: 2 -name: "gb300-fp4-qwen3.5_8k1k_maxtpt_2" - -model: - path: "qwen3.5-fp4" - container: "dynamo-sglang" - precision: "fp4" - -dynamo: - source: - pypi: "1.1.0" - -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 2 - nginx_container: nginx - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 7 - workers: 7 - env: - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_ENABLE_SPEC_V2: "1" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_ENABLE_JIT_DEEPGEMM: "true" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: "cutlass" - FLASHINFER_DISABLE_VERSION_CHECK: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - - args: - served-model-name: "nvidia/Qwen3.5-397B-A17B-NVFP4-V2" - model-path: "/model/" - trust-remote-code: true - - quantization: "modelopt_fp4" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - disaggregation-transfer-backend: "mooncake" - - mem-fraction-static: 0.8 - max-total-tokens: 128000 - chunked-prefill-size: 65536 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - log-level: "info" - page-size: 64 - - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - linear-attn-decode-backend: "flashinfer" - - decode: - nodes: 4 - workers: 1 - - env: - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" - SGLANG_ENABLE_SPEC_V2: "1" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - NCCL_NVLS_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_ENABLE_JIT_DEEPGEMM: "true" - SGLANG_ENABLE_FLASHINFER_GEMM: "true" - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: "cutlass" - SGLANG_MOE_NVFP4_DISPATCH: "1" - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: "1" - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: "1" - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: "1" - FLASHINFER_DISABLE_VERSION_CHECK: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "1024" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "nvidia/Qwen3.5-397B-A17B-NVFP4-V2" - model-path: "/model/" - trust-remote-code: true - - quantization: "modelopt_fp4" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - enable-dp-attention: true - enable-dp-lm-head: true - enable-two-batch-overlap: true - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - disaggregation-transfer-backend: "mooncake" - - chunked-prefill-size: 5120 - max-mamba-cache-size: 5120 - max-total-tokens: 3200000 - max-running-requests: 5120 - mem-fraction-static: 0.8 - watchdog-timeout: 1000000 - page-size: 64 - - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_cutedsl" - moe-a2a-backend: "flashinfer" - disable-shared-experts-fusion: true - linear-attn-decode-backend: "flashinfer" - - decode-log-interval: 50 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - concurrencies: "5120" - req_rate: "inf" - random_range_ratio: 0.8 - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/variants.yaml new file mode 100644 index 0000000000..49ff83da53 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/8k1k/variants.yaml @@ -0,0 +1,338 @@ +# srt-slurm recipes for qwen3.5/sglang/gb300-fp4/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/8k1k/variants.yaml:override_. +# +# + +schema: 2 + +base: + model: + path: qwen3.5-fp4 + container: dynamo-sglang + precision: fp4 + dynamo: + source: + pypi: 1.1.0 + frontend: + type: dynamo + enable_multiple_frontends: true + num_additional_frontends: 2 + nginx_container: nginx + resources: + gpu_type: gb300 + gpus_per_node: 4 + engine: sglang + roles: + prefill: + env: + NO_COLOR: '1' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + SGLANG_ENABLE_SPEC_V2: '1' + PYTHONUNBUFFERED: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + MC_FORCE_MNNVL: '1' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + FLASHINFER_DISABLE_VERSION_CHECK: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + model-path: /model/ + trust-remote-code: true + tensor-parallel-size: 4 + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + mamba-scheduler-strategy: no_buffer + mamba-ssm-dtype: bfloat16 + mamba-track-interval: 2048 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + disaggregation-mode: prefill + disable-radix-cache: true + mem-fraction-static: 0.8 + max-total-tokens: 128000 + load-balance-method: round_robin + page-size: 64 + watchdog-timeout: 1000000 + log-level: info + decode: + workers: 1 + env: + NO_COLOR: '1' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + SGLANG_ENABLE_SPEC_V2: '1' + PYTHONUNBUFFERED: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + MC_FORCE_MNNVL: '1' + MC_TE_METRIC: 'true' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + FLASHINFER_DISABLE_VERSION_CHECK: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + mamba-scheduler-strategy: no_buffer + mamba-ssm-dtype: bfloat16 + mamba-track-interval: 128 + attention-backend: trtllm_mha + linear-attn-decode-backend: flashinfer + disaggregation-mode: decode + disable-radix-cache: true + mem-fraction-static: 0.8 + page-size: 64 + watchdog-timeout: 1000000 + decode-log-interval: 50 + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + random_range_ratio: 0.8 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 19401 + +# Qwen3.5-397B-A17B-NVFP4-V2 Disaggregated 1P1D: TP4 Prefill + TP4 Decode +# Pure tensor parallel, no expert parallel (STP) +# 8k1k sa-bench concurrency sweep on GB300 +# Values taken from ni_experiment_config of the +# sa-qwen-3.5-8k1k-fp4-baseline-low-latency study, row +# qwen3.5-1p_tp4x1d_tp4-aligned-ccsweep (CSV pareto export 2026-06-05). +# (telemetry.dcgm_exporter.port) 9401 is already bound by the cluster-level exporter on im-gb300 nodes; +# (telemetry.dcgm_exporter.port) use a port outside that range. +override_disagg_1p1d_p_tp4_d_tp4_b128_c1x4x8x16x32x64x256_stp: + name: gb300-fp4-qwen3.5_8k1k_lowlat_0 + roles: + prefill: + nodes: 1 + workers: 1 + args: + data-parallel-size: 1 + expert-parallel-size: 1 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + fp4-gemm-backend: flashinfer_cutlass + mm-attention-backend: triton_attn + context-length: 9236 + max-running-requests: 128 + cuda-graph-max-bs: 4 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + scheduler-recv-interval: 10 + stream-interval: 30 + decode: + nodes: 1 + args: + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + fp4-gemm-backend: flashinfer_cutlass + mm-attention-backend: triton_attn + moe-runner-backend: flashinfer_trtllm + context-length: 9236 + max-total-tokens: 1500000 + max-mamba-cache-size: 256 + max-running-requests: 128 + cuda-graph-max-bs: 256 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + scheduler-recv-interval: 10 + stream-interval: 30 + benchmark: + concurrencies: 1x4x8x16x32x64x256 + +# Qwen3.5-397B-A17B-NVFP4-V2 Disaggregated 5P1D wide-EP +# Prefill: 5 workers @ TP4/DP4/EP4 with DP-attn (per-node, DEP4) +# Decode: 1 worker @ TP16/DP16/EP16 with DP-attn + TBO (DEP16, 4 nodes) +# Total: 36 GB300 GPUs (5*4 + 4*4); 8k1k concurrency 1024/2048/3072. +# Values taken from ni_experiment_config of pareto row +# qwen3.5-dep16-fia2a-tbo-cc1024x2048x3072-dynamo-tot-nixl +# (sa-qwen-3.5-8k1k-fp4-baseline-mid-pareto study). +override_disagg_5p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b4096_c2048_stp: + name: gb300-fp4-qwen3.5_8k1k_maxtpt_0 + roles: + prefill: + nodes: 5 + workers: 5 + env: + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_ENABLE_NIXL: '1' + args: + data-parallel-size: 4 + expert-parallel-size: 4 + chunked-prefill-size: 65536 + enable-dp-attention: true + enable-dp-lm-head: true + disaggregation-bootstrap-port: 31000 + disaggregation-transfer-backend: nixl + decode: + nodes: 4 + env: + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '1024' + SGLANG_ENABLE_NIXL: '1' + args: + tensor-parallel-size: 16 + data-parallel-size: 16 + expert-parallel-size: 16 + moe-runner-backend: flashinfer_cutedsl + max-total-tokens: 2200000 + max-mamba-cache-size: 4096 + max-running-requests: 4096 + chunked-prefill-size: 4096 + stream-interval: 50 + enable-dp-attention: true + enable-dp-lm-head: true + enable-two-batch-overlap: true + disaggregation-bootstrap-port: 31000 + disaggregation-transfer-backend: nixl + moe-a2a-backend: flashinfer + disable-shared-experts-fusion: true + benchmark: + concurrencies: '2048' + +# Decode: 1 worker @ TP16/DP16/EP16 with DP-attn + TBO (DEP16, 4 nodes) +# Values taken from ni_experiment_config of pareto row +# (sa-qwen-3.5-8k1k-fp4-baseline-mid-pareto study). +# Qwen3.5-397B-A17B-NVFP4-V2 Disaggregated 6P1D wide-EP +# Prefill: 6 workers @ TP4/DP4/EP4 with DP-attn (per-node, DEP4) +# Total: 40 GB300 GPUs (6*4 + 4*4); 8k1k concurrency 5120. +# qwen3.5-6p_dep4x1d_dep16-fia2a-tbo-cc5120-dynamo-tot-mooncake +override_disagg_6p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b5120_c5120_stp: + name: gb300-fp4-qwen3.5_8k1k_maxtpt_1 + roles: + prefill: + nodes: 6 + workers: 6 + env: + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + args: + data-parallel-size: 4 + expert-parallel-size: 4 + chunked-prefill-size: 65536 + enable-dp-attention: true + enable-dp-lm-head: true + disaggregation-bootstrap-port: 31000 + disaggregation-transfer-backend: mooncake + decode: + nodes: 4 + env: + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '1024' + args: + tensor-parallel-size: 16 + data-parallel-size: 16 + expert-parallel-size: 16 + moe-runner-backend: flashinfer_cutedsl + max-total-tokens: 3200000 + max-mamba-cache-size: 5120 + max-running-requests: 5120 + chunked-prefill-size: 5120 + stream-interval: 50 + enable-dp-attention: true + enable-dp-lm-head: true + enable-two-batch-overlap: true + disaggregation-bootstrap-port: 31000 + disaggregation-transfer-backend: mooncake + moe-a2a-backend: flashinfer + disable-shared-experts-fusion: true + benchmark: + concurrencies: '5120' + +# Decode: 1 worker @ TP16/DP16/EP16 with DP-attn + TBO (DEP16, 4 nodes) +# Values taken from ni_experiment_config of pareto row +# (sa-qwen-3.5-8k1k-fp4-baseline-mid-pareto study). +# Qwen3.5-397B-A17B-NVFP4-V2 Disaggregated 7P1D wide-EP +# Prefill: 7 workers @ TP4/DP4/EP4 with DP-attn (per-node, DEP4) +# Total: 44 GB300 GPUs (7*4 + 4*4); 8k1k concurrency 5120. +# qwen3.5-7p_dep4x1d_dep16-fia2a-tbo-cc5120-dynamo-tot-mooncake +override_disagg_7p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b5120_c5120_stp: + name: gb300-fp4-qwen3.5_8k1k_maxtpt_2 + roles: + prefill: + nodes: 7 + workers: 7 + env: + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + args: + data-parallel-size: 4 + expert-parallel-size: 4 + chunked-prefill-size: 65536 + enable-dp-attention: true + enable-dp-lm-head: true + disaggregation-bootstrap-port: 31000 + disaggregation-transfer-backend: mooncake + decode: + nodes: 4 + env: + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '1024' + args: + tensor-parallel-size: 16 + data-parallel-size: 16 + expert-parallel-size: 16 + moe-runner-backend: flashinfer_cutedsl + max-total-tokens: 3200000 + max-mamba-cache-size: 5120 + max-running-requests: 5120 + chunked-prefill-size: 5120 + stream-interval: 50 + enable-dp-attention: true + enable-dp-lm-head: true + enable-two-batch-overlap: true + disaggregation-bootstrap-port: 31000 + disaggregation-transfer-backend: mooncake + moe-a2a-backend: flashinfer + disable-shared-experts-fusion: true + benchmark: + concurrencies: '5120' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c1-mtp-hicache-jid2530006.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c1-mtp-hicache-jid2530006.yaml deleted file mode 100644 index 1e6f6493fe..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c1-mtp-hicache-jid2530006.yaml +++ /dev/null @@ -1,87 +0,0 @@ -schema: 2 -name: agg-gb300-tp2-c1-mtp-hicache-jid2530006 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 2 -frontend: - type: sglang-router -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 2 - env: - NCCL_NVLS_ENABLE: '1' - PYTHONNOUSERSITE: '1' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - TORCH_CUDA_ARCH_LIST: '10.0' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 2 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-symm-mem: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 8192 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs: 80 - max-running-requests: 72 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.75 - max-mamba-cache-size: 360 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - tokenizer-worker-num: 6 - page-size: 64 - enable-hierarchical-cache: true - hicache-ratio: 0.9 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-write-policy: write_back - mamba-max-states-per-path: 1 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'false' - TP: '2' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c24-mtp-hicache-jid2530012.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c24-mtp-hicache-jid2530012.yaml deleted file mode 100644 index b6579ff4ac..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c24-mtp-hicache-jid2530012.yaml +++ /dev/null @@ -1,87 +0,0 @@ -schema: 2 -name: agg-gb300-tp2-c24-mtp-hicache-jid2530012 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 2 -frontend: - type: sglang-router -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 2 - env: - NCCL_NVLS_ENABLE: '1' - PYTHONNOUSERSITE: '1' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - TORCH_CUDA_ARCH_LIST: '10.0' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 2 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-symm-mem: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 8192 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs: 80 - max-running-requests: 72 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.75 - max-mamba-cache-size: 360 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - tokenizer-worker-num: 6 - page-size: 64 - enable-hierarchical-cache: true - hicache-ratio: 0.9 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-write-policy: write_back - mamba-max-states-per-path: 1 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'false' - TP: '2' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c32-mtp-hicache-jid2530013.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c32-mtp-hicache-jid2530013.yaml deleted file mode 100644 index 0f5a27a447..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c32-mtp-hicache-jid2530013.yaml +++ /dev/null @@ -1,87 +0,0 @@ -schema: 2 -name: agg-gb300-tp2-c32-mtp-hicache-jid2530013 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 2 -frontend: - type: sglang-router -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 2 - env: - NCCL_NVLS_ENABLE: '1' - PYTHONNOUSERSITE: '1' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - TORCH_CUDA_ARCH_LIST: '10.0' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 2 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-symm-mem: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 8192 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs: 80 - max-running-requests: 72 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.75 - max-mamba-cache-size: 360 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - tokenizer-worker-num: 6 - page-size: 64 - enable-hierarchical-cache: true - hicache-ratio: 0.9 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-write-policy: write_back - mamba-max-states-per-path: 1 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'false' - TP: '2' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c40-mtp-hicache-jid2530015.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c40-mtp-hicache-jid2530015.yaml deleted file mode 100644 index 9b3d398808..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c40-mtp-hicache-jid2530015.yaml +++ /dev/null @@ -1,87 +0,0 @@ -schema: 2 -name: agg-gb300-tp2-c40-mtp-hicache-jid2530015 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 2 -frontend: - type: sglang-router -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 2 - env: - NCCL_NVLS_ENABLE: '1' - PYTHONNOUSERSITE: '1' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - TORCH_CUDA_ARCH_LIST: '10.0' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 2 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-symm-mem: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 8192 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs: 80 - max-running-requests: 72 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.75 - max-mamba-cache-size: 360 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - tokenizer-worker-num: 6 - page-size: 64 - enable-hierarchical-cache: true - hicache-ratio: 0.9 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-write-policy: write_back - mamba-max-states-per-path: 1 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'false' - TP: '2' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c44-b1-mtp-hicache-nightly-c20260831.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c44-b1-mtp-hicache-nightly-c20260831.yaml deleted file mode 100644 index 7a5a8417a6..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c44-b1-mtp-hicache-nightly-c20260831.yaml +++ /dev/null @@ -1,132 +0,0 @@ -schema: 2 -name: agg-gb300-tp2-c44-b1-mtp-hicache-nightly-20260831 -model: - path: qwen3.5-fp4 - container: lmsysorg/sglang:nightly-dev-cu13-20260831-bb5e6198 - precision: fp4 -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 2 -frontend: - type: sglang-router - enable_multiple_frontends: false -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 2 - env: - SGLANG_TRTLLM_MHA_DECODE_SEQ_LEN_SPLITS: '1' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - NO_COLOR: '1' - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_NVLS_ENABLE: '0' - MC_FORCE_MNNVL: '1' - NVSHMEM_REMOTE_TRANSPORT: none - MC_TE_METRIC: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_JIT_DEEPGEMM: 'true' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - FLASHINFER_DISABLE_VERSION_CHECK: '1' - SGLANG_DG_CACHE_DIR: /tmp/agentx-pareto-v2/main09ec-pr36248-20260829-tp2-fi-trtllm-c44-mrr1/deepgemm-cache - FLASHINFER_WORKSPACE_BASE: /tmp/agentx-pareto-v2/main09ec-pr36248-20260829-tp2-fi-trtllm-c44-mrr1/flashinfer-cache - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_HEALTH_CHECK_TIMEOUT: '1800' - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' - SGLANG_CACHE_DIR: /tmp/agentx-pareto-v2/main09ec-pr36248-20260829-tp2-fi-trtllm-c44-mrr1/sglang-cache - SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' - SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK: '1' - SGLANG_SCHEDULER_SKIP_ALL_GATHER: '1' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 2 - pipeline-parallel-size: 1 - data-parallel-size: 1 - expert-parallel-size: 1 - moe-dense-tp-size: 2 - enable-dp-attention: false - enable-dp-lm-head: false - moe-a2a-backend: none - load-balance-method: round_robin - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 8192 - mamba-max-states-per-path: 3 - mamba-ssm-dtype: bfloat16 - max-mamba-cache-size: 1536 - context-length: 262144 - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: triton - disable-shared-experts-fusion: true - speculative-algorithm: NEXTN - speculative-num-steps: 6 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - speculative-moe-runner-backend: flashinfer_trtllm - speculative-moe-a2a-backend: none - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - prefill-decode-interval: 0 - mem-fraction-static: 0.85 - max-running-requests: 1 - pp-max-micro-batch-size: 1 - prefill-max-requests: 1 - stream-interval: 20 - decode-log-interval: 10 - watchdog-timeout: 1000000 - weight-loader-prefetch-checkpoints: true - weight-loader-prefetch-num-threads: 4 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-size: 32 - enable-metrics: true - enable-linear-replayssm-spec: true - disable-attn-tp-gather: true - cuda-graph-max-bs-decode: 1 - cuda-graph-bs-decode: - - 1 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'false' - TP: '2' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c44-b2-mtp-hicache-nightly-c20260831.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c44-b2-mtp-hicache-nightly-c20260831.yaml deleted file mode 100644 index ad5b4c86e7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c44-b2-mtp-hicache-nightly-c20260831.yaml +++ /dev/null @@ -1,135 +0,0 @@ -schema: 2 -name: agg-gb300-tp2-c44-b2-mtp-hicache-nightly-20260831 -model: - path: qwen3.5-fp4 - container: lmsysorg/sglang:nightly-dev-cu13-20260831-bb5e6198 - precision: fp4 -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 2 -frontend: - type: sglang-router - enable_multiple_frontends: false -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 2 - env: - SGLANG_TRTLLM_MHA_DECODE_SEQ_LEN_SPLITS: '1' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - NO_COLOR: '1' - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_NVLS_ENABLE: '0' - MC_FORCE_MNNVL: '1' - NVSHMEM_REMOTE_TRANSPORT: none - MC_TE_METRIC: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_JIT_DEEPGEMM: 'true' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - FLASHINFER_DISABLE_VERSION_CHECK: '1' - SGLANG_DG_CACHE_DIR: - /tmp/agentx-pareto-v2/main09ec-pr36248-20260829-tp2-fi-trtllm-hicache128-c44-mrr2/deepgemm-cache - FLASHINFER_WORKSPACE_BASE: - /tmp/agentx-pareto-v2/main09ec-pr36248-20260829-tp2-fi-trtllm-hicache128-c44-mrr2/flashinfer-cache - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_HEALTH_CHECK_TIMEOUT: '1800' - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' - SGLANG_CACHE_DIR: /tmp/agentx-pareto-v2/main09ec-pr36248-20260829-tp2-fi-trtllm-hicache128-c44-mrr2/sglang-cache - SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' - SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK: '1' - SGLANG_SCHEDULER_SKIP_ALL_GATHER: '1' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 2 - pipeline-parallel-size: 1 - data-parallel-size: 1 - expert-parallel-size: 1 - moe-dense-tp-size: 2 - enable-dp-attention: false - enable-dp-lm-head: false - moe-a2a-backend: none - load-balance-method: round_robin - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 8192 - mamba-max-states-per-path: 3 - mamba-ssm-dtype: bfloat16 - max-mamba-cache-size: 1536 - context-length: 262144 - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: triton - disable-shared-experts-fusion: true - speculative-algorithm: NEXTN - speculative-num-steps: 6 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - speculative-moe-runner-backend: flashinfer_trtllm - speculative-moe-a2a-backend: none - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - prefill-decode-interval: 0 - mem-fraction-static: 0.85 - max-running-requests: 2 - pp-max-micro-batch-size: 2 - prefill-max-requests: 2 - cuda-graph-max-bs-decode: 2 - cuda-graph-bs-decode: - - 1 - - 2 - stream-interval: 20 - decode-log-interval: 10 - watchdog-timeout: 1000000 - weight-loader-prefetch-checkpoints: true - weight-loader-prefetch-num-threads: 4 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-size: 128 - enable-metrics: true - enable-linear-replayssm-spec: true - disable-attn-tp-gather: true -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'false' - TP: '2' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c48-mtp-hicache-jid2530017.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c48-mtp-hicache-jid2530017.yaml deleted file mode 100644 index 755cc77d0e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c48-mtp-hicache-jid2530017.yaml +++ /dev/null @@ -1,87 +0,0 @@ -schema: 2 -name: agg-gb300-tp2-c48-mtp-hicache-jid2530017 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 2 -frontend: - type: sglang-router -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 2 - env: - NCCL_NVLS_ENABLE: '1' - PYTHONNOUSERSITE: '1' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - TORCH_CUDA_ARCH_LIST: '10.0' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 2 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-symm-mem: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 8192 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs: 80 - max-running-requests: 72 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.75 - max-mamba-cache-size: 360 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - tokenizer-worker-num: 6 - page-size: 64 - enable-hierarchical-cache: true - hicache-ratio: 0.9 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-write-policy: write_back - mamba-max-states-per-path: 1 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'false' - TP: '2' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c52-mtp-hicache-jid2527406.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c52-mtp-hicache-jid2527406.yaml deleted file mode 100644 index 24a388391d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c52-mtp-hicache-jid2527406.yaml +++ /dev/null @@ -1,87 +0,0 @@ -schema: 2 -name: agg-gb300-tp2-c52-mtp-hicache-jid2527406 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 2 -frontend: - type: sglang-router -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 2 - env: - NCCL_NVLS_ENABLE: '1' - PYTHONNOUSERSITE: '1' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - TORCH_CUDA_ARCH_LIST: '10.0' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 2 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-symm-mem: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 8192 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs: 80 - max-running-requests: 72 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.75 - max-mamba-cache-size: 360 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - tokenizer-worker-num: 6 - page-size: 64 - enable-hierarchical-cache: true - hicache-ratio: 0.9 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-write-policy: write_back - mamba-max-states-per-path: 1 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'false' - TP: '2' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c64-mtp-hicache-jid2527410.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c64-mtp-hicache-jid2527410.yaml deleted file mode 100644 index ee3999af7b..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c64-mtp-hicache-jid2527410.yaml +++ /dev/null @@ -1,87 +0,0 @@ -schema: 2 -name: agg-gb300-tp2-c64-mtp-hicache-jid2527410 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 2 -frontend: - type: sglang-router -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 2 - env: - NCCL_NVLS_ENABLE: '1' - PYTHONNOUSERSITE: '1' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGL_ENABLE_JIT_DEEPGEMM: 'false' - TORCH_CUDA_ARCH_LIST: '10.0' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - tensor-parallel-size: 2 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-symm-mem: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 8192 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - cuda-graph-max-bs: 80 - max-running-requests: 72 - max-prefill-tokens: 16384 - chunked-prefill-size: 16384 - mem-fraction-static: 0.75 - max-mamba-cache-size: 360 - allow-auto-truncate: true - stream-interval: 50 - scheduler-recv-interval: 10 - tokenizer-worker-num: 6 - page-size: 64 - enable-hierarchical-cache: true - hicache-ratio: 0.9 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-write-policy: write_back - mamba-max-states-per-path: 1 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'false' - TP: '2' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp8-c7-b1-mtp-hicache-nightly-c20260831.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp8-c7-b1-mtp-hicache-nightly-c20260831.yaml deleted file mode 100644 index bc672a8342..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp8-c7-b1-mtp-hicache-nightly-c20260831.yaml +++ /dev/null @@ -1,132 +0,0 @@ -schema: 2 -name: agg-gb300-tp8-c7-b1-mtp-hicache-nightly-20260831 -model: - path: qwen3.5-fp4 - container: lmsysorg/sglang:nightly-dev-cu13-20260831-bb5e6198 - precision: fp4 -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -frontend: - type: sglang-router - enable_multiple_frontends: false -engine: sglang -roles: - agg: - nodes: 2 - workers: 1 - gpus: 8 - env: - SGLANG_TRTLLM_MHA_DECODE_SEQ_LEN_SPLITS: '1' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - NO_COLOR: '1' - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_NVLS_ENABLE: '0' - MC_FORCE_MNNVL: '1' - NVSHMEM_REMOTE_TRANSPORT: none - MC_TE_METRIC: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_JIT_DEEPGEMM: 'true' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - FLASHINFER_DISABLE_VERSION_CHECK: '1' - SGLANG_DG_CACHE_DIR: /tmp/agentx-pareto-v2/c7-mrr1/deepgemm-cache - FLASHINFER_WORKSPACE_BASE: /tmp/agentx-pareto-v2/c7-mrr1/flashinfer-cache - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_HEALTH_CHECK_TIMEOUT: '1800' - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' - SGLANG_CACHE_DIR: /tmp/agentx-pareto-v2/c7-mrr1/sglang-cache - SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' - SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK: '1' - SGLANG_SCHEDULER_SKIP_ALL_GATHER: '1' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 8 - pipeline-parallel-size: 1 - data-parallel-size: 1 - expert-parallel-size: 1 - moe-dense-tp-size: 8 - enable-dp-attention: false - enable-dp-lm-head: false - moe-a2a-backend: none - load-balance-method: round_robin - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 8192 - mamba-max-states-per-path: 3 - mamba-ssm-dtype: bfloat16 - max-mamba-cache-size: 1536 - context-length: 262144 - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_cutedsl - linear-attn-decode-backend: triton - disable-shared-experts-fusion: true - speculative-algorithm: NEXTN - speculative-num-steps: 6 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - speculative-moe-runner-backend: flashinfer_cutedsl - speculative-moe-a2a-backend: none - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - mem-fraction-static: 0.85 - max-running-requests: 1 - pp-max-micro-batch-size: 1 - prefill-max-requests: 1 - cuda-graph-max-bs-decode: 1 - cuda-graph-bs-decode: - - 1 - disable-prefill-cuda-graph: true - stream-interval: 20 - decode-log-interval: 10 - watchdog-timeout: 1000000 - weight-loader-prefetch-checkpoints: true - weight-loader-prefetch-num-threads: 4 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-size: 32 - enable-metrics: true - enable-linear-replayssm-spec: true - disable-attn-tp-gather: true -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - TP: '8' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp2-tp2-c72-mtp-hicache-session-jid2527415.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp2-tp2-c72-mtp-hicache-session-jid2527415.yaml deleted file mode 100644 index 3580fbbe29..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp2-tp2-c72-mtp-hicache-session-jid2527415.yaml +++ /dev/null @@ -1,211 +0,0 @@ -schema: 2 -name: disagg-gb300-1p1d-tp2-tp2-c72-mtp-hicache-session-jid2527415 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -dynamo: - install: true - source: - rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 2 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 8 -frontend: - type: dynamo - nginx_session_affinity: true - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - trust-remote-code: true - tensor-parallel-size: 2 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - mamba-ssm-dtype: bfloat16 - disaggregation-mode: prefill - disaggregation-bootstrap-port: 31000 - load-balance-method: round_robin - watchdog-timeout: 1000000 - log-level: info - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - mem-fraction-static: 0.8 - chunked-prefill-size: 65536 - disable-cuda-graph: true - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 2048 - enable-hierarchical-cache: true - hicache-write-policy: write_back - max-mamba-cache-size: 320 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-ratio: 0.9 - mamba-max-states-per-path: 1 - decode: - nodes: 1 - workers: 1 - gpus: 2 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' - SGLANG_ENABLE_JIT_DEEPGEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' - SGLANG_HEALTH_CHECK_TIMEOUT: '1800' - SGLANG_HEALTH_STARTING_OK: '1' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 2 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - prefill-round-robin-balance: true - mamba-scheduler-strategy: no_buffer - mamba-track-interval: 128 - mamba-ssm-dtype: bfloat16 - disaggregation-mode: decode - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - disable-shared-experts-fusion: true - ep-dispatch-algorithm: static - eplb-algorithm: deepseek - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - chunked-prefill-size: 4096 - mem-fraction-static: 0.75 - max-mamba-cache-size: 80 - max-running-requests: 80 - cuda-graph-max-bs: 128 - watchdog-timeout: 1000000 - decode-log-interval: 10 - mamba-max-states-per-path: -1 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c128-mtp-hicache-session-jid2527417.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c128-mtp-hicache-session-jid2527417.yaml deleted file mode 100644 index 908f37b284..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c128-mtp-hicache-session-jid2527417.yaml +++ /dev/null @@ -1,206 +0,0 @@ -schema: 2 -name: disagg-gb300-1p1d-tp4-tp4-c128-mtp-hicache-session-jid2527417 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -dynamo: - install: true - source: - rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 8 -frontend: - type: dynamo - nginx_session_affinity: true - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - trust-remote-code: true - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - mamba-ssm-dtype: bfloat16 - disaggregation-mode: prefill - disaggregation-bootstrap-port: 31000 - load-balance-method: round_robin - watchdog-timeout: 1000000 - log-level: info - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - mem-fraction-static: 0.8 - chunked-prefill-size: 65536 - disable-cuda-graph: true - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 2048 - enable-hierarchical-cache: true - hicache-write-policy: write_back - mamba-max-states-per-path: 3 - max-mamba-cache-size: 1100 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-ratio: 0.9 - decode: - nodes: 1 - workers: 1 - gpus: 4 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' - SGLANG_ENABLE_JIT_DEEPGEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' - SGLANG_HEALTH_CHECK_TIMEOUT: '1800' - SGLANG_HEALTH_STARTING_OK: '1' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - prefill-round-robin-balance: true - mamba-scheduler-strategy: no_buffer - mamba-track-interval: 128 - mamba-ssm-dtype: bfloat16 - disaggregation-mode: decode - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - disable-shared-experts-fusion: true - ep-dispatch-algorithm: static - eplb-algorithm: deepseek - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - chunked-prefill-size: 4096 - mem-fraction-static: 0.75 - max-mamba-cache-size: 1024 - max-running-requests: 512 - cuda-graph-max-bs: 160 - watchdog-timeout: 1000000 - decode-log-interval: 10 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c16-mtp-hicache-session-jid2530027.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c16-mtp-hicache-session-jid2530027.yaml deleted file mode 100644 index a9e4f40a9a..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c16-mtp-hicache-session-jid2530027.yaml +++ /dev/null @@ -1,206 +0,0 @@ -schema: 2 -name: disagg-gb300-1p1d-tp4-tp4-c16-mtp-hicache-session-jid2530027 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -dynamo: - install: true - source: - rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 8 -frontend: - type: dynamo - nginx_session_affinity: true - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - trust-remote-code: true - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - mamba-ssm-dtype: bfloat16 - disaggregation-mode: prefill - disaggregation-bootstrap-port: 31000 - load-balance-method: round_robin - watchdog-timeout: 1000000 - log-level: info - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - mem-fraction-static: 0.8 - chunked-prefill-size: 65536 - disable-cuda-graph: true - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 2048 - enable-hierarchical-cache: true - hicache-write-policy: write_back - mamba-max-states-per-path: 3 - max-mamba-cache-size: 1100 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-ratio: 0.9 - decode: - nodes: 1 - workers: 1 - gpus: 4 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' - SGLANG_ENABLE_JIT_DEEPGEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' - SGLANG_HEALTH_CHECK_TIMEOUT: '1800' - SGLANG_HEALTH_STARTING_OK: '1' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - prefill-round-robin-balance: true - mamba-scheduler-strategy: no_buffer - mamba-track-interval: 128 - mamba-ssm-dtype: bfloat16 - disaggregation-mode: decode - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - disable-shared-experts-fusion: true - ep-dispatch-algorithm: static - eplb-algorithm: deepseek - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - chunked-prefill-size: 4096 - mem-fraction-static: 0.75 - max-mamba-cache-size: 1024 - max-running-requests: 512 - cuda-graph-max-bs: 64 - watchdog-timeout: 1000000 - decode-log-interval: 10 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c32-mtp-hicache-session-jid2530028.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c32-mtp-hicache-session-jid2530028.yaml deleted file mode 100644 index 3854d0ad20..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c32-mtp-hicache-session-jid2530028.yaml +++ /dev/null @@ -1,206 +0,0 @@ -schema: 2 -name: disagg-gb300-1p1d-tp4-tp4-c32-mtp-hicache-session-jid2530028 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -dynamo: - install: true - source: - rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 8 -frontend: - type: dynamo - nginx_session_affinity: true - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - trust-remote-code: true - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - mamba-ssm-dtype: bfloat16 - disaggregation-mode: prefill - disaggregation-bootstrap-port: 31000 - load-balance-method: round_robin - watchdog-timeout: 1000000 - log-level: info - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - mem-fraction-static: 0.8 - chunked-prefill-size: 65536 - disable-cuda-graph: true - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 2048 - enable-hierarchical-cache: true - hicache-write-policy: write_back - mamba-max-states-per-path: 3 - max-mamba-cache-size: 1100 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-ratio: 0.9 - decode: - nodes: 1 - workers: 1 - gpus: 4 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' - SGLANG_ENABLE_JIT_DEEPGEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' - SGLANG_HEALTH_CHECK_TIMEOUT: '1800' - SGLANG_HEALTH_STARTING_OK: '1' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - prefill-round-robin-balance: true - mamba-scheduler-strategy: no_buffer - mamba-track-interval: 128 - mamba-ssm-dtype: bfloat16 - disaggregation-mode: decode - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - disable-shared-experts-fusion: true - ep-dispatch-algorithm: static - eplb-algorithm: deepseek - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - chunked-prefill-size: 4096 - mem-fraction-static: 0.75 - max-mamba-cache-size: 1024 - max-running-requests: 512 - cuda-graph-max-bs: 64 - watchdog-timeout: 1000000 - decode-log-interval: 10 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c64-mtp-hicache-session-jid2530029.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c64-mtp-hicache-session-jid2530029.yaml deleted file mode 100644 index d8325d5792..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c64-mtp-hicache-session-jid2530029.yaml +++ /dev/null @@ -1,206 +0,0 @@ -schema: 2 -name: disagg-gb300-1p1d-tp4-tp4-c64-mtp-hicache-session-jid2530029 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -dynamo: - install: true - source: - rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 8 -frontend: - type: dynamo - nginx_session_affinity: true - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - trust-remote-code: true - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - mamba-ssm-dtype: bfloat16 - disaggregation-mode: prefill - disaggregation-bootstrap-port: 31000 - load-balance-method: round_robin - watchdog-timeout: 1000000 - log-level: info - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - mem-fraction-static: 0.8 - chunked-prefill-size: 65536 - disable-cuda-graph: true - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 2048 - enable-hierarchical-cache: true - hicache-write-policy: write_back - mamba-max-states-per-path: 3 - max-mamba-cache-size: 1100 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-ratio: 0.9 - decode: - nodes: 1 - workers: 1 - gpus: 4 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' - SGLANG_ENABLE_JIT_DEEPGEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' - SGLANG_HEALTH_CHECK_TIMEOUT: '1800' - SGLANG_HEALTH_STARTING_OK: '1' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - prefill-round-robin-balance: true - mamba-scheduler-strategy: no_buffer - mamba-track-interval: 128 - mamba-ssm-dtype: bfloat16 - disaggregation-mode: decode - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - disable-shared-experts-fusion: true - ep-dispatch-algorithm: static - eplb-algorithm: deepseek - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - chunked-prefill-size: 4096 - mem-fraction-static: 0.75 - max-mamba-cache-size: 1024 - max-running-requests: 512 - cuda-graph-max-bs: 80 - watchdog-timeout: 1000000 - decode-log-interval: 10 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c8-mtp-hicache-session-jid2530030.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c8-mtp-hicache-session-jid2530030.yaml deleted file mode 100644 index 8a776aa9af..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c8-mtp-hicache-session-jid2530030.yaml +++ /dev/null @@ -1,206 +0,0 @@ -schema: 2 -name: disagg-gb300-1p1d-tp4-tp4-c8-mtp-hicache-session-jid2530030 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -dynamo: - install: true - source: - rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 8 -frontend: - type: dynamo - nginx_session_affinity: true - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - trust-remote-code: true - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - mamba-ssm-dtype: bfloat16 - disaggregation-mode: prefill - disaggregation-bootstrap-port: 31000 - load-balance-method: round_robin - watchdog-timeout: 1000000 - log-level: info - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - mem-fraction-static: 0.8 - chunked-prefill-size: 65536 - disable-cuda-graph: true - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 2048 - enable-hierarchical-cache: true - hicache-write-policy: write_back - mamba-max-states-per-path: 3 - max-mamba-cache-size: 1100 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-ratio: 0.9 - decode: - nodes: 1 - workers: 1 - gpus: 4 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' - SGLANG_ENABLE_JIT_DEEPGEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' - SGLANG_HEALTH_CHECK_TIMEOUT: '1800' - SGLANG_HEALTH_STARTING_OK: '1' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - prefill-round-robin-balance: true - mamba-scheduler-strategy: no_buffer - mamba-track-interval: 128 - mamba-ssm-dtype: bfloat16 - disaggregation-mode: decode - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - disable-shared-experts-fusion: true - ep-dispatch-algorithm: static - eplb-algorithm: deepseek - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - chunked-prefill-size: 4096 - mem-fraction-static: 0.75 - max-mamba-cache-size: 1024 - max-running-requests: 512 - cuda-graph-max-bs: 64 - watchdog-timeout: 1000000 - decode-log-interval: 10 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c96-mtp-hicache-session-jid2527409.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c96-mtp-hicache-session-jid2527409.yaml deleted file mode 100644 index c0ff95a0f7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c96-mtp-hicache-session-jid2527409.yaml +++ /dev/null @@ -1,206 +0,0 @@ -schema: 2 -name: disagg-gb300-1p1d-tp4-tp4-c96-mtp-hicache-session-jid2527409 -model: - path: qwen3.5-fp4 - container: dynamo-sglang - precision: fp4 -dynamo: - install: true - source: - rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 8 -frontend: - type: dynamo - nginx_session_affinity: true - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - nginx_session_affinity_header: X-Dynamo-Session-ID -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - trust-remote-code: true - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - mamba-ssm-dtype: bfloat16 - disaggregation-mode: prefill - disaggregation-bootstrap-port: 31000 - load-balance-method: round_robin - watchdog-timeout: 1000000 - log-level: info - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - mem-fraction-static: 0.8 - chunked-prefill-size: 65536 - disable-cuda-graph: true - mamba-scheduler-strategy: extra_buffer - mamba-track-interval: 2048 - enable-hierarchical-cache: true - hicache-write-policy: write_back - mamba-max-states-per-path: 3 - max-mamba-cache-size: 1100 - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-ratio: 0.9 - decode: - nodes: 1 - workers: 1 - gpus: 4 - env: - FLASHINFER_DISABLE_VERSION_CHECK: '1' - FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache - MC_FORCE_MNNVL: '1' - MC_TE_METRIC: 'true' - NCCL_CUMEM_ENABLE: '1' - NCCL_MNNVL_ENABLE: '1' - NCCL_NVLS_ENABLE: '1' - NO_COLOR: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' - SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_DISAGG_STAGING_BUFFER: '1' - SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' - SGLANG_ENABLE_JIT_DEEPGEMM: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' - SGLANG_HEALTH_CHECK_TIMEOUT: '1800' - SGLANG_HEALTH_STARTING_OK: '1' - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-cache-report: true - enable-metrics: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - prefill-round-robin-balance: true - mamba-scheduler-strategy: no_buffer - mamba-track-interval: 128 - mamba-ssm-dtype: bfloat16 - disaggregation-mode: decode - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_trtllm - linear-attn-decode-backend: flashinfer - disable-shared-experts-fusion: true - ep-dispatch-algorithm: static - eplb-algorithm: deepseek - speculative-algorithm: NEXTN - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - chunked-prefill-size: 4096 - mem-fraction-static: 0.75 - max-mamba-cache-size: 1024 - max-running-requests: 512 - cuda-graph-max-bs: 128 - watchdog-timeout: 1000000 - decode-log-interval: 10 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-3p2d-pp4-dep4-c704-mtp-hicache-nightly-c20260831.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-3p2d-pp4-dep4-c704-mtp-hicache-nightly-c20260831.yaml deleted file mode 100644 index e790701576..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-3p2d-pp4-dep4-c704-mtp-hicache-nightly-c20260831.yaml +++ /dev/null @@ -1,250 +0,0 @@ -schema: 2 -name: disagg-gb300-3p2d-pp4-dep4-c704-mtp-hicache-nightly-20260831 -model: - path: qwen3.5-fp4 - container: lmsysorg/sglang:nightly-dev-cu13-20260831-bb5e6198 - precision: fp4 -dynamo: - install: true - source: - rev: 61c37bc33c3e88da3facaa39fa2b5caf616740ba -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 8 -frontend: - type: dynamo - nginx_session_affinity: true - enable_multiple_frontends: true - num_additional_frontends: 1 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - router-replica-sync: true - router-queue-threshold: None - router-temperature: '1.0' -engine: sglang -roles: - prefill: - nodes: 3 - workers: 3 - gpus: 4 - env: - # The 0831 image contains the PP+spec runtime merged by SGLang #35758, - # but still carries the pre-merge PP+spec assert in validation_hook.py. - PYTHONOPTIMIZE: '1' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - SGLANG_DISAGG_STAGING_BUFFER: '1' - NO_COLOR: '1' - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_NVLS_ENABLE: '0' - MC_FORCE_MNNVL: '1' - NVSHMEM_REMOTE_TRANSPORT: none - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - FLASHINFER_DISABLE_VERSION_CHECK: '1' - SGLANG_DG_CACHE_DIR: /tmp/agentx-upstream-main/prefill-deepgemm-cache - FLASHINFER_WORKSPACE_BASE: /tmp/agentx-upstream-main/prefill-flashinfer-cache - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK: '1' - SGLANG_FLASHINFER_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '8192' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_PP_LAYER_PARTITION: 16,16,16,12 - SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' - SGLANG_FLASHINFER_AUTOTUNE_EXTEND: '1' - SGLANG_CACHE_DIR: /tmp/agentx-upstream-main/prefill-sglang-cache - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 1 - pipeline-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - enable-symm-mem: false - disable-overlap-schedule: true - enable-dynamic-chunking: false - mamba-ssm-dtype: bfloat16 - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 1048576 - mamba-max-states-per-path: 3 - max-mamba-cache-size: 1536 - max-running-requests: 128 - disaggregation-mode: prefill - disaggregation-bootstrap-port: 31000 - load-balance-method: round_robin - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_cutedsl - speculative-algorithm: NEXTN - speculative-num-steps: 5 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 6 - linear-attn-decode-backend: triton - mem-fraction-static: 0.85 - max-prefill-tokens: 32768 - chunked-prefill-size: 32768 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-ratio: 0.9 - watchdog-timeout: 1000000 - log-level: info - nccl-port: 29500 - scheduler-recv-interval: 1 - weight-loader-prefetch-checkpoints: true - weight-loader-prefetch-num-threads: 4 - moe-dense-tp-size: 1 - disable-shared-experts-fusion: true - moe-a2a-backend: none - speculative-moe-a2a-backend: none - speculative-moe-runner-backend: flashinfer_trtllm - pp-async-batch-depth: 1 - disable-cuda-graph: true - decode: - nodes: 2 - workers: 2 - gpus: 4 - env: - SGLANG_USE_SYMM_MEM_DP_SYNC: 'true' - SGLANG_TRTLLM_MHA_DECODE_SEQ_LEN_SPLITS: '4' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - SGLANG_DISAGG_STAGING_BUFFER: '1' - NO_COLOR: '1' - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_NVLS_ENABLE: '0' - MC_FORCE_MNNVL: '1' - NVSHMEM_REMOTE_TRANSPORT: none - MC_TE_METRIC: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_JIT_DEEPGEMM: 'true' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - FLASHINFER_DISABLE_VERSION_CHECK: '1' - SGLANG_DG_CACHE_DIR: /tmp/agentx-upstream-main/decode-deepgemm-cache - FLASHINFER_WORKSPACE_BASE: /tmp/agentx-upstream-main/decode-flashinfer-cache - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_HEALTH_CHECK_TIMEOUT: '1800' - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_CACHE_DIR: /tmp/agentx-upstream-main/decode-sglang-cache - SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - pipeline-parallel-size: 1 - data-parallel-size: 4 - expert-parallel-size: 4 - moe-dense-tp-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: round_robin - mamba-scheduler-strategy: no_buffer - mamba-track-interval: 128 - mamba-ssm-dtype: bfloat16 - disaggregation-mode: decode - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_cutedsl - linear-attn-decode-backend: triton - disable-shared-experts-fusion: true - moe-a2a-backend: flashinfer - ep-dispatch-algorithm: static - eplb-algorithm: deepseek - speculative-algorithm: NEXTN - speculative-num-steps: 5 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 6 - speculative-draft-model-quantization: nvfp4_online - speculative-moe-runner-backend: flashinfer_trtllm_routed - speculative-moe-a2a-backend: flashinfer - chunked-prefill-size: 4096 - mem-fraction-static: 0.7 - max-mamba-cache-size: 468 - max-running-requests: 320 - cuda-graph-max-bs: 80 - disaggregation-decode-extra-slots: 2 - stream-interval: 30 - enable-linear-replayssm-spec: true - linear-replayssm-cache-len: 32 - decode-log-interval: 30 - watchdog-timeout: 1000000 - weight-loader-prefetch-checkpoints: true - weight-loader-prefetch-num-threads: 4 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-3p4d-pp4-dep4-c565-mtp-hicache-nightly-c20260831.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-3p4d-pp4-dep4-c565-mtp-hicache-nightly-c20260831.yaml deleted file mode 100644 index 9146587211..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-3p4d-pp4-dep4-c565-mtp-hicache-nightly-c20260831.yaml +++ /dev/null @@ -1,250 +0,0 @@ -schema: 2 -name: disagg-gb300-3p4d-pp4-dep4-c565-mtp-hicache-nightly-20260831 -model: - path: qwen3.5-fp4 - container: lmsysorg/sglang:nightly-dev-cu13-20260831-bb5e6198 - precision: fp4 -dynamo: - install: true - source: - rev: 61c37bc33c3e88da3facaa39fa2b5caf616740ba -slurm: - time_limit: '8:00:00' -health_check: - max_attempts: 1440 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 8 -frontend: - type: dynamo - nginx_session_affinity: true - enable_multiple_frontends: true - num_additional_frontends: 1 - env: - PIP_BREAK_SYSTEM_PACKAGES: '1' - args: - router-mode: kv - router-session-affinity-ttl-secs: '3600' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - router-replica-sync: true - router-queue-threshold: None - router-temperature: '1.0' -engine: sglang -roles: - prefill: - nodes: 3 - workers: 3 - gpus: 4 - env: - # The 0831 image contains the PP+spec runtime merged by SGLang #35758, - # but still carries the pre-merge PP+spec assert in validation_hook.py. - PYTHONOPTIMIZE: '1' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - SGLANG_DISAGG_STAGING_BUFFER: '1' - NO_COLOR: '1' - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_NVLS_ENABLE: '0' - MC_FORCE_MNNVL: '1' - NVSHMEM_REMOTE_TRANSPORT: none - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - FLASHINFER_DISABLE_VERSION_CHECK: '1' - SGLANG_DG_CACHE_DIR: /tmp/agentx-followup/c565-cutedsl-g3072/prefill-deepgemm-cache - FLASHINFER_WORKSPACE_BASE: /tmp/agentx-followup/c565-cutedsl-g3072/prefill-flashinfer-cache - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK: '1' - SGLANG_FLASHINFER_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '8192' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_PP_LAYER_PARTITION: 16,16,16,12 - SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' - SGLANG_FLASHINFER_AUTOTUNE_EXTEND: '1' - SGLANG_CACHE_DIR: /tmp/agentx-followup/c565-cutedsl-g3072/prefill-sglang-cache - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 1 - pipeline-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - enable-dp-attention: false - enable-dp-lm-head: false - enable-symm-mem: false - disable-overlap-schedule: true - enable-dynamic-chunking: false - mamba-ssm-dtype: bfloat16 - mamba-radix-cache-strategy: extra_buffer - mamba-track-interval: 1048576 - mamba-max-states-per-path: 3 - max-mamba-cache-size: 1536 - max-running-requests: 128 - disaggregation-mode: prefill - disaggregation-bootstrap-port: 31000 - load-balance-method: round_robin - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_cutedsl - speculative-algorithm: NEXTN - speculative-num-steps: 4 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 5 - linear-attn-decode-backend: triton - mem-fraction-static: 0.85 - max-prefill-tokens: 32768 - chunked-prefill-size: 32768 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-io-backend: kernel - hicache-mem-layout: page_first_direct - hicache-ratio: 0.9 - watchdog-timeout: 1000000 - log-level: info - nccl-port: 29500 - scheduler-recv-interval: 1 - weight-loader-prefetch-checkpoints: true - weight-loader-prefetch-num-threads: 4 - moe-dense-tp-size: 1 - disable-shared-experts-fusion: true - moe-a2a-backend: none - speculative-moe-a2a-backend: none - speculative-moe-runner-backend: flashinfer_trtllm - pp-async-batch-depth: 1 - disable-cuda-graph: true - decode: - nodes: 4 - workers: 4 - gpus: 4 - env: - SGLANG_USE_SYMM_MEM_DP_SYNC: 'true' - SGLANG_TRTLLM_MHA_DECODE_SEQ_LEN_SPLITS: '4' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - SGLANG_DISAGG_STAGING_BUFFER: '1' - NO_COLOR: '1' - PYTHONUNBUFFERED: '1' - PIP_BREAK_SYSTEM_PACKAGES: '1' - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' - NCCL_MNNVL_ENABLE: '1' - NCCL_CUMEM_ENABLE: '1' - NCCL_NVLS_ENABLE: '0' - MC_FORCE_MNNVL: '1' - NVSHMEM_REMOTE_TRANSPORT: none - MC_TE_METRIC: 'true' - SGLANG_ENABLE_SPEC_V2: '1' - SGLANG_ENABLE_FLASHINFER_GEMM: 'true' - SGLANG_ENABLE_JIT_DEEPGEMM: 'true' - SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass - SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - FLASHINFER_DISABLE_VERSION_CHECK: '1' - SGLANG_DG_CACHE_DIR: /tmp/agentx-followup/c565-cutedsl-g3072/decode-deepgemm-cache - FLASHINFER_WORKSPACE_BASE: /tmp/agentx-followup/c565-cutedsl-g3072/decode-flashinfer-cache - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' - SGLANG_HEALTH_CHECK_TIMEOUT: '1800' - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' - SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' - SGLANG_CACHE_DIR: /tmp/agentx-followup/c565-cutedsl-g3072/decode-sglang-cache - SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - quantization: modelopt_fp4 - kv-cache-dtype: fp8_e4m3 - tensor-parallel-size: 4 - pipeline-parallel-size: 1 - data-parallel-size: 4 - expert-parallel-size: 4 - moe-dense-tp-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: round_robin - mamba-scheduler-strategy: no_buffer - mamba-track-interval: 128 - mamba-ssm-dtype: bfloat16 - disaggregation-mode: decode - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - page-size: 64 - attention-backend: trtllm_mha - moe-runner-backend: flashinfer_cutedsl - linear-attn-decode-backend: triton - disable-shared-experts-fusion: true - moe-a2a-backend: flashinfer - ep-dispatch-algorithm: static - eplb-algorithm: deepseek - speculative-algorithm: NEXTN - speculative-num-steps: 4 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 5 - speculative-draft-model-quantization: nvfp4_online - speculative-moe-runner-backend: flashinfer_trtllm_routed - speculative-moe-a2a-backend: flashinfer - chunked-prefill-size: 4096 - mem-fraction-static: 0.7 - max-mamba-cache-size: 468 - max-running-requests: 320 - cuda-graph-max-bs: 80 - disaggregation-decode-extra-slots: 2 - stream-interval: 30 - enable-linear-replayssm-spec: true - linear-replayssm-cache-len: 32 - decode-log-interval: 30 - watchdog-timeout: 1000000 - weight-loader-prefetch-checkpoints: true - weight-loader-prefetch-num-threads: 4 -sbatch_directives: - mem: '0' - cpus-per-task: '144' -srun_options: - mem: '0' - container-remap-root: '' -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..bd204319ad --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml @@ -0,0 +1,2513 @@ +# srt-slurm recipes for qwen3.5/sglang/gb300-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: qwen3.5-fp4 + precision: fp4 + slurm: + time_limit: '8:00:00' + health_check: + max_attempts: 1440 + interval_seconds: 10 + resources: + gpu_type: gb300 + frontend: {} + engine: sglang + roles: {} + sbatch_directives: + mem: '0' + cpus-per-task: '144' + srun_options: + mem: '0' + container-remap-root: '' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'sglang:' + +override_agg_tp2_c1_mtp_hicache_jid2530006: + name: agg-gb300-tp2-c1-mtp-hicache-jid2530006 + model: + container: dynamo-sglang + resources: + gpus_per_node: 2 + frontend: + type: sglang-router + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + env: + NCCL_NVLS_ENABLE: '1' + PYTHONNOUSERSITE: '1' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + TORCH_CUDA_ARCH_LIST: '10.0' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + tensor-parallel-size: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 8192 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + cuda-graph-max-bs: 80 + max-running-requests: 72 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.75 + max-mamba-cache-size: 360 + allow-auto-truncate: true + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + page-size: 64 + enable-hierarchical-cache: true + hicache-ratio: 0.9 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-write-policy: write_back + mamba-max-states-per-path: 1 + benchmark: + env: + IS_MULTINODE: 'false' + TP: '2' + +override_agg_tp2_c24_mtp_hicache_jid2530012: + name: agg-gb300-tp2-c24-mtp-hicache-jid2530012 + model: + container: dynamo-sglang + resources: + gpus_per_node: 2 + frontend: + type: sglang-router + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + env: + NCCL_NVLS_ENABLE: '1' + PYTHONNOUSERSITE: '1' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + TORCH_CUDA_ARCH_LIST: '10.0' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + tensor-parallel-size: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 8192 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + cuda-graph-max-bs: 80 + max-running-requests: 72 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.75 + max-mamba-cache-size: 360 + allow-auto-truncate: true + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + page-size: 64 + enable-hierarchical-cache: true + hicache-ratio: 0.9 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-write-policy: write_back + mamba-max-states-per-path: 1 + benchmark: + env: + IS_MULTINODE: 'false' + TP: '2' + +override_agg_tp2_c32_mtp_hicache_jid2530013: + name: agg-gb300-tp2-c32-mtp-hicache-jid2530013 + model: + container: dynamo-sglang + resources: + gpus_per_node: 2 + frontend: + type: sglang-router + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + env: + NCCL_NVLS_ENABLE: '1' + PYTHONNOUSERSITE: '1' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + TORCH_CUDA_ARCH_LIST: '10.0' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + tensor-parallel-size: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 8192 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + cuda-graph-max-bs: 80 + max-running-requests: 72 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.75 + max-mamba-cache-size: 360 + allow-auto-truncate: true + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + page-size: 64 + enable-hierarchical-cache: true + hicache-ratio: 0.9 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-write-policy: write_back + mamba-max-states-per-path: 1 + benchmark: + env: + IS_MULTINODE: 'false' + TP: '2' + +override_agg_tp2_c40_mtp_hicache_jid2530015: + name: agg-gb300-tp2-c40-mtp-hicache-jid2530015 + model: + container: dynamo-sglang + resources: + gpus_per_node: 2 + frontend: + type: sglang-router + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + env: + NCCL_NVLS_ENABLE: '1' + PYTHONNOUSERSITE: '1' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + TORCH_CUDA_ARCH_LIST: '10.0' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + tensor-parallel-size: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 8192 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + cuda-graph-max-bs: 80 + max-running-requests: 72 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.75 + max-mamba-cache-size: 360 + allow-auto-truncate: true + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + page-size: 64 + enable-hierarchical-cache: true + hicache-ratio: 0.9 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-write-policy: write_back + mamba-max-states-per-path: 1 + benchmark: + env: + IS_MULTINODE: 'false' + TP: '2' + +override_agg_tp2_c44_b1_mtp_hicache_nightly_c20260831: + name: agg-gb300-tp2-c44-b1-mtp-hicache-nightly-20260831 + model: + container: lmsysorg/sglang:nightly-dev-cu13-20260831-bb5e6198 + resources: + gpus_per_node: 2 + frontend: + type: sglang-router + enable_multiple_frontends: false + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + env: + SGLANG_TRTLLM_MHA_DECODE_SEQ_LEN_SPLITS: '1' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + NO_COLOR: '1' + PYTHONUNBUFFERED: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_NVLS_ENABLE: '0' + MC_FORCE_MNNVL: '1' + NVSHMEM_REMOTE_TRANSPORT: none + MC_TE_METRIC: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + FLASHINFER_DISABLE_VERSION_CHECK: '1' + SGLANG_DG_CACHE_DIR: /tmp/agentx-pareto-v2/main09ec-pr36248-20260829-tp2-fi-trtllm-c44-mrr1/deepgemm-cache + FLASHINFER_WORKSPACE_BASE: /tmp/agentx-pareto-v2/main09ec-pr36248-20260829-tp2-fi-trtllm-c44-mrr1/flashinfer-cache + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + SGLANG_CACHE_DIR: /tmp/agentx-pareto-v2/main09ec-pr36248-20260829-tp2-fi-trtllm-c44-mrr1/sglang-cache + SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' + SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK: '1' + SGLANG_SCHEDULER_SKIP_ALL_GATHER: '1' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 2 + pipeline-parallel-size: 1 + data-parallel-size: 1 + expert-parallel-size: 1 + moe-dense-tp-size: 2 + enable-dp-attention: false + enable-dp-lm-head: false + moe-a2a-backend: none + load-balance-method: round_robin + mamba-radix-cache-strategy: extra_buffer + mamba-track-interval: 8192 + mamba-max-states-per-path: 3 + mamba-ssm-dtype: bfloat16 + max-mamba-cache-size: 1536 + context-length: 262144 + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: triton + disable-shared-experts-fusion: true + speculative-algorithm: NEXTN + speculative-num-steps: 6 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + speculative-moe-runner-backend: flashinfer_trtllm + speculative-moe-a2a-backend: none + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + prefill-decode-interval: 0 + mem-fraction-static: 0.85 + max-running-requests: 1 + pp-max-micro-batch-size: 1 + prefill-max-requests: 1 + stream-interval: 20 + decode-log-interval: 10 + watchdog-timeout: 1000000 + weight-loader-prefetch-checkpoints: true + weight-loader-prefetch-num-threads: 4 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-size: 32 + enable-metrics: true + enable-linear-replayssm-spec: true + disable-attn-tp-gather: true + cuda-graph-max-bs-decode: 1 + cuda-graph-bs-decode: [1] + benchmark: + env: + IS_MULTINODE: 'false' + TP: '2' + +override_agg_tp2_c44_b2_mtp_hicache_nightly_c20260831: + name: agg-gb300-tp2-c44-b2-mtp-hicache-nightly-20260831 + model: + container: lmsysorg/sglang:nightly-dev-cu13-20260831-bb5e6198 + resources: + gpus_per_node: 2 + frontend: + type: sglang-router + enable_multiple_frontends: false + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + env: + SGLANG_TRTLLM_MHA_DECODE_SEQ_LEN_SPLITS: '1' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + NO_COLOR: '1' + PYTHONUNBUFFERED: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_NVLS_ENABLE: '0' + MC_FORCE_MNNVL: '1' + NVSHMEM_REMOTE_TRANSPORT: none + MC_TE_METRIC: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + FLASHINFER_DISABLE_VERSION_CHECK: '1' + SGLANG_DG_CACHE_DIR: /tmp/agentx-pareto-v2/main09ec-pr36248-20260829-tp2-fi-trtllm-hicache128-c44-mrr2/deepgemm-cache + FLASHINFER_WORKSPACE_BASE: /tmp/agentx-pareto-v2/main09ec-pr36248-20260829-tp2-fi-trtllm-hicache128-c44-mrr2/flashinfer-cache + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + SGLANG_CACHE_DIR: /tmp/agentx-pareto-v2/main09ec-pr36248-20260829-tp2-fi-trtllm-hicache128-c44-mrr2/sglang-cache + SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' + SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK: '1' + SGLANG_SCHEDULER_SKIP_ALL_GATHER: '1' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 2 + pipeline-parallel-size: 1 + data-parallel-size: 1 + expert-parallel-size: 1 + moe-dense-tp-size: 2 + enable-dp-attention: false + enable-dp-lm-head: false + moe-a2a-backend: none + load-balance-method: round_robin + mamba-radix-cache-strategy: extra_buffer + mamba-track-interval: 8192 + mamba-max-states-per-path: 3 + mamba-ssm-dtype: bfloat16 + max-mamba-cache-size: 1536 + context-length: 262144 + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: triton + disable-shared-experts-fusion: true + speculative-algorithm: NEXTN + speculative-num-steps: 6 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + speculative-moe-runner-backend: flashinfer_trtllm + speculative-moe-a2a-backend: none + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + prefill-decode-interval: 0 + mem-fraction-static: 0.85 + max-running-requests: 2 + pp-max-micro-batch-size: 2 + prefill-max-requests: 2 + cuda-graph-max-bs-decode: 2 + cuda-graph-bs-decode: [1, 2] + stream-interval: 20 + decode-log-interval: 10 + watchdog-timeout: 1000000 + weight-loader-prefetch-checkpoints: true + weight-loader-prefetch-num-threads: 4 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-size: 128 + enable-metrics: true + enable-linear-replayssm-spec: true + disable-attn-tp-gather: true + benchmark: + env: + IS_MULTINODE: 'false' + TP: '2' + +override_agg_tp2_c48_mtp_hicache_jid2530017: + name: agg-gb300-tp2-c48-mtp-hicache-jid2530017 + model: + container: dynamo-sglang + resources: + gpus_per_node: 2 + frontend: + type: sglang-router + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + env: + NCCL_NVLS_ENABLE: '1' + PYTHONNOUSERSITE: '1' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + TORCH_CUDA_ARCH_LIST: '10.0' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + tensor-parallel-size: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 8192 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + cuda-graph-max-bs: 80 + max-running-requests: 72 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.75 + max-mamba-cache-size: 360 + allow-auto-truncate: true + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + page-size: 64 + enable-hierarchical-cache: true + hicache-ratio: 0.9 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-write-policy: write_back + mamba-max-states-per-path: 1 + benchmark: + env: + IS_MULTINODE: 'false' + TP: '2' + +override_agg_tp2_c52_mtp_hicache_jid2527406: + name: agg-gb300-tp2-c52-mtp-hicache-jid2527406 + model: + container: dynamo-sglang + resources: + gpus_per_node: 2 + frontend: + type: sglang-router + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + env: + NCCL_NVLS_ENABLE: '1' + PYTHONNOUSERSITE: '1' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + TORCH_CUDA_ARCH_LIST: '10.0' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + tensor-parallel-size: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 8192 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + cuda-graph-max-bs: 80 + max-running-requests: 72 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.75 + max-mamba-cache-size: 360 + allow-auto-truncate: true + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + page-size: 64 + enable-hierarchical-cache: true + hicache-ratio: 0.9 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-write-policy: write_back + mamba-max-states-per-path: 1 + benchmark: + env: + IS_MULTINODE: 'false' + TP: '2' + +override_agg_tp2_c64_mtp_hicache_jid2527410: + name: agg-gb300-tp2-c64-mtp-hicache-jid2527410 + model: + container: dynamo-sglang + resources: + gpus_per_node: 2 + frontend: + type: sglang-router + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + env: + NCCL_NVLS_ENABLE: '1' + PYTHONNOUSERSITE: '1' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + TORCH_CUDA_ARCH_LIST: '10.0' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + tensor-parallel-size: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 8192 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + cuda-graph-max-bs: 80 + max-running-requests: 72 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.75 + max-mamba-cache-size: 360 + allow-auto-truncate: true + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + page-size: 64 + enable-hierarchical-cache: true + hicache-ratio: 0.9 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-write-policy: write_back + mamba-max-states-per-path: 1 + benchmark: + env: + IS_MULTINODE: 'false' + TP: '2' + +override_agg_tp8_c7_b1_mtp_hicache_nightly_c20260831: + name: agg-gb300-tp8-c7-b1-mtp-hicache-nightly-20260831 + model: + container: lmsysorg/sglang:nightly-dev-cu13-20260831-bb5e6198 + resources: + gpus_per_node: 4 + frontend: + type: sglang-router + enable_multiple_frontends: false + roles: + agg: + nodes: 2 + workers: 1 + gpus: 8 + env: + SGLANG_TRTLLM_MHA_DECODE_SEQ_LEN_SPLITS: '1' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + NO_COLOR: '1' + PYTHONUNBUFFERED: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_NVLS_ENABLE: '0' + MC_FORCE_MNNVL: '1' + NVSHMEM_REMOTE_TRANSPORT: none + MC_TE_METRIC: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + FLASHINFER_DISABLE_VERSION_CHECK: '1' + SGLANG_DG_CACHE_DIR: /tmp/agentx-pareto-v2/c7-mrr1/deepgemm-cache + FLASHINFER_WORKSPACE_BASE: /tmp/agentx-pareto-v2/c7-mrr1/flashinfer-cache + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + SGLANG_CACHE_DIR: /tmp/agentx-pareto-v2/c7-mrr1/sglang-cache + SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' + SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK: '1' + SGLANG_SCHEDULER_SKIP_ALL_GATHER: '1' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 8 + pipeline-parallel-size: 1 + data-parallel-size: 1 + expert-parallel-size: 1 + moe-dense-tp-size: 8 + enable-dp-attention: false + enable-dp-lm-head: false + moe-a2a-backend: none + load-balance-method: round_robin + mamba-radix-cache-strategy: extra_buffer + mamba-track-interval: 8192 + mamba-max-states-per-path: 3 + mamba-ssm-dtype: bfloat16 + max-mamba-cache-size: 1536 + context-length: 262144 + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_cutedsl + linear-attn-decode-backend: triton + disable-shared-experts-fusion: true + speculative-algorithm: NEXTN + speculative-num-steps: 6 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + speculative-moe-runner-backend: flashinfer_cutedsl + speculative-moe-a2a-backend: none + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + mem-fraction-static: 0.85 + max-running-requests: 1 + pp-max-micro-batch-size: 1 + prefill-max-requests: 1 + cuda-graph-max-bs-decode: 1 + cuda-graph-bs-decode: [1] + disable-prefill-cuda-graph: true + stream-interval: 20 + decode-log-interval: 10 + watchdog-timeout: 1000000 + weight-loader-prefetch-checkpoints: true + weight-loader-prefetch-num-threads: 4 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-size: 32 + enable-metrics: true + enable-linear-replayssm-spec: true + disable-attn-tp-gather: true + benchmark: + env: + IS_MULTINODE: 'true' + TP: '8' + +override_disagg_1p1d_tp2_tp2_c72_mtp_hicache_session_jid2527415: + name: disagg-gb300-1p1d-tp2-tp2-c72-mtp-hicache-session-jid2527415 + model: + container: dynamo-sglang + resources: + gpus_per_node: 2 + frontend: + type: dynamo + enable_multiple_frontends: true + nginx_session_affinity: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + nginx_session_affinity_header: X-Dynamo-Session-ID + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 2 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + trust-remote-code: true + tensor-parallel-size: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + mamba-ssm-dtype: bfloat16 + disaggregation-mode: prefill + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + log-level: info + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + mem-fraction-static: 0.8 + chunked-prefill-size: 65536 + disable-cuda-graph: true + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 2048 + enable-hierarchical-cache: true + hicache-write-policy: write_back + max-mamba-cache-size: 320 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-ratio: 0.9 + mamba-max-states-per-path: 1 + decode: + nodes: 1 + workers: 1 + gpus: 2 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + prefill-round-robin-balance: true + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + mamba-ssm-dtype: bfloat16 + disaggregation-mode: decode + disable-radix-cache: true + disaggregation-bootstrap-port: 31000 + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + disable-shared-experts-fusion: true + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + chunked-prefill-size: 4096 + mem-fraction-static: 0.75 + max-mamba-cache-size: 80 + max-running-requests: 80 + cuda-graph-max-bs: 128 + watchdog-timeout: 1000000 + decode-log-interval: 10 + mamba-max-states-per-path: -1 + benchmark: + env: + IS_MULTINODE: 'true' + dynamo: + install: true + source: + rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 8 + +override_disagg_1p1d_tp4_tp4_c128_mtp_hicache_session_jid2527417: + name: disagg-gb300-1p1d-tp4-tp4-c128-mtp-hicache-session-jid2527417 + model: + container: dynamo-sglang + resources: + gpus_per_node: 4 + frontend: + type: dynamo + enable_multiple_frontends: true + nginx_session_affinity: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + nginx_session_affinity_header: X-Dynamo-Session-ID + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + mamba-ssm-dtype: bfloat16 + disaggregation-mode: prefill + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + log-level: info + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + mem-fraction-static: 0.8 + chunked-prefill-size: 65536 + disable-cuda-graph: true + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 2048 + enable-hierarchical-cache: true + hicache-write-policy: write_back + mamba-max-states-per-path: 3 + max-mamba-cache-size: 1100 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-ratio: 0.9 + decode: + nodes: 1 + workers: 1 + gpus: 4 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + prefill-round-robin-balance: true + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + mamba-ssm-dtype: bfloat16 + disaggregation-mode: decode + disable-radix-cache: true + disaggregation-bootstrap-port: 31000 + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + disable-shared-experts-fusion: true + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + chunked-prefill-size: 4096 + mem-fraction-static: 0.75 + max-mamba-cache-size: 1024 + max-running-requests: 512 + cuda-graph-max-bs: 160 + watchdog-timeout: 1000000 + decode-log-interval: 10 + benchmark: + env: + IS_MULTINODE: 'true' + dynamo: + install: true + source: + rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 8 + +override_disagg_1p1d_tp4_tp4_c16_mtp_hicache_session_jid2530027: + name: disagg-gb300-1p1d-tp4-tp4-c16-mtp-hicache-session-jid2530027 + model: + container: dynamo-sglang + resources: + gpus_per_node: 4 + frontend: + type: dynamo + enable_multiple_frontends: true + nginx_session_affinity: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + nginx_session_affinity_header: X-Dynamo-Session-ID + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + mamba-ssm-dtype: bfloat16 + disaggregation-mode: prefill + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + log-level: info + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + mem-fraction-static: 0.8 + chunked-prefill-size: 65536 + disable-cuda-graph: true + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 2048 + enable-hierarchical-cache: true + hicache-write-policy: write_back + mamba-max-states-per-path: 3 + max-mamba-cache-size: 1100 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-ratio: 0.9 + decode: + nodes: 1 + workers: 1 + gpus: 4 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + prefill-round-robin-balance: true + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + mamba-ssm-dtype: bfloat16 + disaggregation-mode: decode + disable-radix-cache: true + disaggregation-bootstrap-port: 31000 + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + disable-shared-experts-fusion: true + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + chunked-prefill-size: 4096 + mem-fraction-static: 0.75 + max-mamba-cache-size: 1024 + max-running-requests: 512 + cuda-graph-max-bs: 64 + watchdog-timeout: 1000000 + decode-log-interval: 10 + benchmark: + env: + IS_MULTINODE: 'true' + dynamo: + install: true + source: + rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 8 + +override_disagg_1p1d_tp4_tp4_c32_mtp_hicache_session_jid2530028: + name: disagg-gb300-1p1d-tp4-tp4-c32-mtp-hicache-session-jid2530028 + model: + container: dynamo-sglang + resources: + gpus_per_node: 4 + frontend: + type: dynamo + enable_multiple_frontends: true + nginx_session_affinity: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + nginx_session_affinity_header: X-Dynamo-Session-ID + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + mamba-ssm-dtype: bfloat16 + disaggregation-mode: prefill + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + log-level: info + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + mem-fraction-static: 0.8 + chunked-prefill-size: 65536 + disable-cuda-graph: true + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 2048 + enable-hierarchical-cache: true + hicache-write-policy: write_back + mamba-max-states-per-path: 3 + max-mamba-cache-size: 1100 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-ratio: 0.9 + decode: + nodes: 1 + workers: 1 + gpus: 4 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + prefill-round-robin-balance: true + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + mamba-ssm-dtype: bfloat16 + disaggregation-mode: decode + disable-radix-cache: true + disaggregation-bootstrap-port: 31000 + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + disable-shared-experts-fusion: true + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + chunked-prefill-size: 4096 + mem-fraction-static: 0.75 + max-mamba-cache-size: 1024 + max-running-requests: 512 + cuda-graph-max-bs: 64 + watchdog-timeout: 1000000 + decode-log-interval: 10 + benchmark: + env: + IS_MULTINODE: 'true' + dynamo: + install: true + source: + rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 8 + +override_disagg_1p1d_tp4_tp4_c64_mtp_hicache_session_jid2530029: + name: disagg-gb300-1p1d-tp4-tp4-c64-mtp-hicache-session-jid2530029 + model: + container: dynamo-sglang + resources: + gpus_per_node: 4 + frontend: + type: dynamo + enable_multiple_frontends: true + nginx_session_affinity: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + nginx_session_affinity_header: X-Dynamo-Session-ID + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + mamba-ssm-dtype: bfloat16 + disaggregation-mode: prefill + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + log-level: info + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + mem-fraction-static: 0.8 + chunked-prefill-size: 65536 + disable-cuda-graph: true + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 2048 + enable-hierarchical-cache: true + hicache-write-policy: write_back + mamba-max-states-per-path: 3 + max-mamba-cache-size: 1100 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-ratio: 0.9 + decode: + nodes: 1 + workers: 1 + gpus: 4 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + prefill-round-robin-balance: true + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + mamba-ssm-dtype: bfloat16 + disaggregation-mode: decode + disable-radix-cache: true + disaggregation-bootstrap-port: 31000 + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + disable-shared-experts-fusion: true + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + chunked-prefill-size: 4096 + mem-fraction-static: 0.75 + max-mamba-cache-size: 1024 + max-running-requests: 512 + cuda-graph-max-bs: 80 + watchdog-timeout: 1000000 + decode-log-interval: 10 + benchmark: + env: + IS_MULTINODE: 'true' + dynamo: + install: true + source: + rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 8 + +override_disagg_1p1d_tp4_tp4_c8_mtp_hicache_session_jid2530030: + name: disagg-gb300-1p1d-tp4-tp4-c8-mtp-hicache-session-jid2530030 + model: + container: dynamo-sglang + resources: + gpus_per_node: 4 + frontend: + type: dynamo + enable_multiple_frontends: true + nginx_session_affinity: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + nginx_session_affinity_header: X-Dynamo-Session-ID + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + mamba-ssm-dtype: bfloat16 + disaggregation-mode: prefill + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + log-level: info + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + mem-fraction-static: 0.8 + chunked-prefill-size: 65536 + disable-cuda-graph: true + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 2048 + enable-hierarchical-cache: true + hicache-write-policy: write_back + mamba-max-states-per-path: 3 + max-mamba-cache-size: 1100 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-ratio: 0.9 + decode: + nodes: 1 + workers: 1 + gpus: 4 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + prefill-round-robin-balance: true + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + mamba-ssm-dtype: bfloat16 + disaggregation-mode: decode + disable-radix-cache: true + disaggregation-bootstrap-port: 31000 + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + disable-shared-experts-fusion: true + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + chunked-prefill-size: 4096 + mem-fraction-static: 0.75 + max-mamba-cache-size: 1024 + max-running-requests: 512 + cuda-graph-max-bs: 64 + watchdog-timeout: 1000000 + decode-log-interval: 10 + benchmark: + env: + IS_MULTINODE: 'true' + dynamo: + install: true + source: + rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 8 + +override_disagg_1p1d_tp4_tp4_c96_mtp_hicache_session_jid2527409: + name: disagg-gb300-1p1d-tp4-tp4-c96-mtp-hicache-session-jid2527409 + model: + container: dynamo-sglang + resources: + gpus_per_node: 4 + frontend: + type: dynamo + enable_multiple_frontends: true + nginx_session_affinity: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + nginx_session_affinity_header: X-Dynamo-Session-ID + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + mamba-ssm-dtype: bfloat16 + disaggregation-mode: prefill + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + log-level: info + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + mem-fraction-static: 0.8 + chunked-prefill-size: 65536 + disable-cuda-graph: true + mamba-scheduler-strategy: extra_buffer + mamba-track-interval: 2048 + enable-hierarchical-cache: true + hicache-write-policy: write_back + mamba-max-states-per-path: 3 + max-mamba-cache-size: 1100 + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-ratio: 0.9 + decode: + nodes: 1 + workers: 1 + gpus: 4 + env: + FLASHINFER_DISABLE_VERSION_CHECK: '1' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + MC_FORCE_MNNVL: '1' + MC_TE_METRIC: 'true' + NCCL_CUMEM_ENABLE: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_NVLS_ENABLE: '1' + NO_COLOR: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '256' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DISAGG_STAGING_BUFFER: '1' + SGLANG_DISAGG_STAGING_BUFFER_SIZE_MB: '1024' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-cache-report: true + enable-metrics: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + prefill-round-robin-balance: true + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + mamba-ssm-dtype: bfloat16 + disaggregation-mode: decode + disable-radix-cache: true + disaggregation-bootstrap-port: 31000 + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + linear-attn-decode-backend: flashinfer + disable-shared-experts-fusion: true + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + speculative-algorithm: NEXTN + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + chunked-prefill-size: 4096 + mem-fraction-static: 0.75 + max-mamba-cache-size: 1024 + max-running-requests: 512 + cuda-graph-max-bs: 128 + watchdog-timeout: 1000000 + decode-log-interval: 10 + benchmark: + env: + IS_MULTINODE: 'true' + dynamo: + install: true + source: + rev: 5a638087d82c990d35c69cb8e41c2c2582e9ef9a + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 8 + +override_disagg_3p2d_pp4_dep4_c704_mtp_hicache_nightly_c20260831: + name: disagg-gb300-3p2d-pp4-dep4-c704-mtp-hicache-nightly-20260831 + model: + container: lmsysorg/sglang:nightly-dev-cu13-20260831-bb5e6198 + resources: + gpus_per_node: 4 + frontend: + type: dynamo + enable_multiple_frontends: true + nginx_session_affinity: true + num_additional_frontends: 1 + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + router-replica-sync: true + router-queue-threshold: None + router-temperature: '1.0' + roles: + prefill: + nodes: 3 + workers: 3 + gpus: 4 + env: + # The 0831 image contains the PP+spec runtime merged by SGLang #35758, + # but still carries the pre-merge PP+spec assert in validation_hook.py. + PYTHONOPTIMIZE: '1' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + SGLANG_DISAGG_STAGING_BUFFER: '1' + NO_COLOR: '1' + PYTHONUNBUFFERED: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_NVLS_ENABLE: '0' + MC_FORCE_MNNVL: '1' + NVSHMEM_REMOTE_TRANSPORT: none + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + FLASHINFER_DISABLE_VERSION_CHECK: '1' + SGLANG_DG_CACHE_DIR: /tmp/agentx-upstream-main/prefill-deepgemm-cache + FLASHINFER_WORKSPACE_BASE: /tmp/agentx-upstream-main/prefill-flashinfer-cache + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK: '1' + SGLANG_FLASHINFER_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '8192' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_PP_LAYER_PARTITION: 16,16,16,12 + SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' + SGLANG_FLASHINFER_AUTOTUNE_EXTEND: '1' + SGLANG_CACHE_DIR: /tmp/agentx-upstream-main/prefill-sglang-cache + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 1 + pipeline-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + enable-symm-mem: false + disable-overlap-schedule: true + enable-dynamic-chunking: false + mamba-ssm-dtype: bfloat16 + mamba-radix-cache-strategy: extra_buffer + mamba-track-interval: 1048576 + mamba-max-states-per-path: 3 + max-mamba-cache-size: 1536 + max-running-requests: 128 + disaggregation-mode: prefill + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_cutedsl + speculative-algorithm: NEXTN + speculative-num-steps: 5 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 6 + linear-attn-decode-backend: triton + mem-fraction-static: 0.85 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-ratio: 0.9 + watchdog-timeout: 1000000 + log-level: info + nccl-port: 29500 + scheduler-recv-interval: 1 + weight-loader-prefetch-checkpoints: true + weight-loader-prefetch-num-threads: 4 + moe-dense-tp-size: 1 + disable-shared-experts-fusion: true + moe-a2a-backend: none + speculative-moe-a2a-backend: none + speculative-moe-runner-backend: flashinfer_trtllm + pp-async-batch-depth: 1 + disable-cuda-graph: true + decode: + nodes: 2 + workers: 2 + gpus: 4 + env: + SGLANG_USE_SYMM_MEM_DP_SYNC: 'true' + SGLANG_TRTLLM_MHA_DECODE_SEQ_LEN_SPLITS: '4' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + SGLANG_DISAGG_STAGING_BUFFER: '1' + NO_COLOR: '1' + PYTHONUNBUFFERED: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_NVLS_ENABLE: '0' + MC_FORCE_MNNVL: '1' + NVSHMEM_REMOTE_TRANSPORT: none + MC_TE_METRIC: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + FLASHINFER_DISABLE_VERSION_CHECK: '1' + SGLANG_DG_CACHE_DIR: /tmp/agentx-upstream-main/decode-deepgemm-cache + FLASHINFER_WORKSPACE_BASE: /tmp/agentx-upstream-main/decode-flashinfer-cache + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_CACHE_DIR: /tmp/agentx-upstream-main/decode-sglang-cache + SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 4 + pipeline-parallel-size: 1 + data-parallel-size: 4 + expert-parallel-size: 4 + moe-dense-tp-size: 1 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: round_robin + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + mamba-ssm-dtype: bfloat16 + disaggregation-mode: decode + disable-radix-cache: true + disaggregation-bootstrap-port: 31000 + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_cutedsl + linear-attn-decode-backend: triton + disable-shared-experts-fusion: true + moe-a2a-backend: flashinfer + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + speculative-algorithm: NEXTN + speculative-num-steps: 5 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 6 + speculative-draft-model-quantization: nvfp4_online + speculative-moe-runner-backend: flashinfer_trtllm_routed + speculative-moe-a2a-backend: flashinfer + chunked-prefill-size: 4096 + mem-fraction-static: 0.7 + max-mamba-cache-size: 468 + max-running-requests: 320 + cuda-graph-max-bs: 80 + disaggregation-decode-extra-slots: 2 + stream-interval: 30 + enable-linear-replayssm-spec: true + linear-replayssm-cache-len: 32 + decode-log-interval: 30 + watchdog-timeout: 1000000 + weight-loader-prefetch-checkpoints: true + weight-loader-prefetch-num-threads: 4 + benchmark: + env: + IS_MULTINODE: 'true' + dynamo: + install: true + source: + rev: 61c37bc33c3e88da3facaa39fa2b5caf616740ba + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 8 + +override_disagg_3p4d_pp4_dep4_c565_mtp_hicache_nightly_c20260831: + name: disagg-gb300-3p4d-pp4-dep4-c565-mtp-hicache-nightly-20260831 + model: + container: lmsysorg/sglang:nightly-dev-cu13-20260831-bb5e6198 + resources: + gpus_per_node: 4 + frontend: + type: dynamo + enable_multiple_frontends: true + nginx_session_affinity: true + num_additional_frontends: 1 + env: + PIP_BREAK_SYSTEM_PACKAGES: '1' + args: + router-mode: kv + router-session-affinity-ttl-secs: '3600' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + router-replica-sync: true + router-queue-threshold: None + router-temperature: '1.0' + roles: + prefill: + nodes: 3 + workers: 3 + gpus: 4 + env: + # The 0831 image contains the PP+spec runtime merged by SGLang #35758, + # but still carries the pre-merge PP+spec assert in validation_hook.py. + PYTHONOPTIMIZE: '1' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + SGLANG_DISAGG_STAGING_BUFFER: '1' + NO_COLOR: '1' + PYTHONUNBUFFERED: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_NVLS_ENABLE: '0' + MC_FORCE_MNNVL: '1' + NVSHMEM_REMOTE_TRANSPORT: none + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + FLASHINFER_DISABLE_VERSION_CHECK: '1' + SGLANG_DG_CACHE_DIR: /tmp/agentx-followup/c565-cutedsl-g3072/prefill-deepgemm-cache + FLASHINFER_WORKSPACE_BASE: /tmp/agentx-followup/c565-cutedsl-g3072/prefill-flashinfer-cache + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK: '1' + SGLANG_FLASHINFER_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '8192' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_PP_LAYER_PARTITION: 16,16,16,12 + SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' + SGLANG_FLASHINFER_AUTOTUNE_EXTEND: '1' + SGLANG_CACHE_DIR: /tmp/agentx-followup/c565-cutedsl-g3072/prefill-sglang-cache + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 1 + pipeline-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-dp-attention: false + enable-dp-lm-head: false + enable-symm-mem: false + disable-overlap-schedule: true + enable-dynamic-chunking: false + mamba-ssm-dtype: bfloat16 + mamba-radix-cache-strategy: extra_buffer + mamba-track-interval: 1048576 + mamba-max-states-per-path: 3 + max-mamba-cache-size: 1536 + max-running-requests: 128 + disaggregation-mode: prefill + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_cutedsl + speculative-algorithm: NEXTN + speculative-num-steps: 4 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 5 + linear-attn-decode-backend: triton + mem-fraction-static: 0.85 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-io-backend: kernel + hicache-mem-layout: page_first_direct + hicache-ratio: 0.9 + watchdog-timeout: 1000000 + log-level: info + nccl-port: 29500 + scheduler-recv-interval: 1 + weight-loader-prefetch-checkpoints: true + weight-loader-prefetch-num-threads: 4 + moe-dense-tp-size: 1 + disable-shared-experts-fusion: true + moe-a2a-backend: none + speculative-moe-a2a-backend: none + speculative-moe-runner-backend: flashinfer_trtllm + pp-async-batch-depth: 1 + disable-cuda-graph: true + decode: + nodes: 4 + workers: 4 + gpus: 4 + env: + SGLANG_USE_SYMM_MEM_DP_SYNC: 'true' + SGLANG_TRTLLM_MHA_DECODE_SEQ_LEN_SPLITS: '4' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + SGLANG_DISAGG_STAGING_BUFFER: '1' + NO_COLOR: '1' + PYTHONUNBUFFERED: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '1800' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + NCCL_NVLS_ENABLE: '0' + MC_FORCE_MNNVL: '1' + NVSHMEM_REMOTE_TRANSPORT: none + MC_TE_METRIC: 'true' + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + SGLANG_ENABLE_JIT_DEEPGEMM: 'true' + SGLANG_FLASHINFER_FP4_GEMM_BACKEND: cutlass + SGLANG_MOE_NVFP4_DISPATCH: '1' + SGLANG_CUTEDSL_MOE_NVFP4_DISPATCH: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + FLASHINFER_DISABLE_VERSION_CHECK: '1' + SGLANG_DG_CACHE_DIR: /tmp/agentx-followup/c565-cutedsl-g3072/decode-deepgemm-cache + FLASHINFER_WORKSPACE_BASE: /tmp/agentx-followup/c565-cutedsl-g3072/decode-flashinfer-cache + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + SGLANG_DISAGG_STAGING_POOL_SIZE_MB: '8192' + SGLANG_CACHE_DIR: /tmp/agentx-followup/c565-cutedsl-g3072/decode-sglang-cache + SGLANG_FLASHINFER_AUTOTUNE_CACHE: '1' + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + tensor-parallel-size: 4 + pipeline-parallel-size: 1 + data-parallel-size: 4 + expert-parallel-size: 4 + moe-dense-tp-size: 1 + enable-dp-attention: true + enable-dp-lm-head: true + load-balance-method: round_robin + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + mamba-ssm-dtype: bfloat16 + disaggregation-mode: decode + disable-radix-cache: true + disaggregation-bootstrap-port: 31000 + page-size: 64 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_cutedsl + linear-attn-decode-backend: triton + disable-shared-experts-fusion: true + moe-a2a-backend: flashinfer + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + speculative-algorithm: NEXTN + speculative-num-steps: 4 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 5 + speculative-draft-model-quantization: nvfp4_online + speculative-moe-runner-backend: flashinfer_trtllm_routed + speculative-moe-a2a-backend: flashinfer + chunked-prefill-size: 4096 + mem-fraction-static: 0.7 + max-mamba-cache-size: 468 + max-running-requests: 320 + cuda-graph-max-bs: 80 + disaggregation-decode-extra-slots: 2 + stream-interval: 30 + enable-linear-replayssm-spec: true + linear-replayssm-cache-len: 32 + decode-log-interval: 30 + watchdog-timeout: 1000000 + weight-loader-prefetch-checkpoints: true + weight-loader-prefetch-num-threads: 4 + benchmark: + env: + IS_MULTINODE: 'true' + dynamo: + install: true + source: + rev: 61c37bc33c3e88da3facaa39fa2b5caf616740ba + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 8 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml deleted file mode 100644 index 9050a22244..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml +++ /dev/null @@ -1,156 +0,0 @@ -# Qwen3.5 FP8 GB300 disaggregated MTP 1P1D TP4/TP4 configuration. - -schema: 2 -name: "qwen3.5-fp8-gb300-mtp-8k1k-1p1d-tp4-tp4" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 1 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - attention-backend: "trtllm_mha" - tensor-parallel-size: 4 - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - moe-runner-backend: "flashinfer_trtllm" - disable-radix-cache: true - max-running-requests: 1024 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - context-length: 16384 - cuda-graph-max-bs: 1024 - decode-log-interval: 1 - stream-interval: 50 - disaggregation-mode: "prefill" - - decode: - nodes: 1 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - attention-backend: "trtllm_mha" - tensor-parallel-size: 4 - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - max-mamba-cache-size: 1024 - moe-runner-backend: "flashinfer_trtllm" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disable-radix-cache: true - max-running-requests: 1024 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - context-length: 16384 - cuda-graph-max-bs: 1024 - decode-log-interval: 1 - stream-interval: 50 - disaggregation-mode: "decode" - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - random_range_ratio: 0.8 - concurrencies: "1x2x8" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml deleted file mode 100644 index 8ce688a8ba..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Qwen3.5 FP8 GB300 disaggregated MTP 1P1D TEP8/TEP8 configuration. - -schema: 2 -name: "qwen3.5-fp8-gb300-mtp-8k1k-1p1d-tep8-tep8" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 1 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 2 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - attention-backend: "trtllm_mha" - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 8 - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - moe-runner-backend: "flashinfer_trtllm" - disable-radix-cache: true - max-running-requests: 1024 - mem-fraction-static: 0.8 - max-total-tokens: 128000 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - context-length: 9236 - cuda-graph-max-bs: 320 - scheduler-recv-interval: 10 - decode-log-interval: 50 - stream-interval: 50 - disaggregation-mode: "prefill" - disaggregation-transfer-backend: "mooncake" - - decode: - nodes: 2 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - attention-backend: "trtllm_mha" - tensor-parallel-size: 8 - data-parallel-size: 1 - expert-parallel-size: 8 - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - max-mamba-cache-size: 1024 - moe-runner-backend: "flashinfer_trtllm" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disable-radix-cache: true - max-running-requests: 1024 - mem-fraction-static: 0.8 - max-total-tokens: 2200000 - chunked-prefill-size: 4096 - max-prefill-tokens: 16384 - context-length: 9236 - cuda-graph-max-bs: 320 - scheduler-recv-interval: 10 - decode-log-interval: 50 - stream-interval: 50 - disaggregation-mode: "decode" - disaggregation-transfer-backend: "mooncake" - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "32x48x80" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-tp4-tp4-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-tp4-tp4-stp.yaml deleted file mode 100644 index b3a68a8e8e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-tp4-tp4-stp.yaml +++ /dev/null @@ -1,136 +0,0 @@ -schema: 2 -name: "qwen3.5-1p1d-tp4-tp4" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - wheel: "1.5.0.dev20260917" -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 1 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 1 - workers: 1 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - MC_FORCE_MNNVL: "1" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - attention-backend: "trtllm_mha" - tensor-parallel-size: 4 - mamba-ssm-dtype: "bfloat16" - moe-runner-backend: "flashinfer_trtllm" - disable-radix-cache: true - max-running-requests: 1024 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - context-length: 16384 - cuda-graph-max-bs-decode: 1024 - decode-log-interval: 1 - stream-interval: 50 - disaggregation-mode: "prefill" - - decode: - nodes: 1 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_HEALTH_CHECK_TIMEOUT: "3600" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - attention-backend: "trtllm_mha" - tensor-parallel-size: 4 - mamba-ssm-dtype: "bfloat16" - moe-runner-backend: "flashinfer_trtllm" - disable-radix-cache: true - max-running-requests: 1024 - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - context-length: 16384 - cuda-graph-max-bs-decode: 1024 - decode-log-interval: 1 - stream-interval: 50 - disaggregation-mode: "decode" - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - random_range_ratio: 0.8 - concurrencies: "1x2x4x8x16x32x64x128" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - # 9401 is already bound by the cluster-level exporter on im-gb300 nodes; - # use a port outside that range. - port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml deleted file mode 100644 index e4588b3dda..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml +++ /dev/null @@ -1,186 +0,0 @@ -# Qwen3.5 FP8 GB300 disaggregated MTP 3P1D DEP4/DEP16 configuration. - -schema: 2 -name: "qwen3.5-fp8-gb300-mtp-8k1k-3p1d-dep4-dep16" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 2 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 3 - workers: 3 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - mem-fraction-static: 0.7 - chunked-prefill-size: 65536 - max-prefill-tokens: 16384 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - moe-dense-tp-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - prefill-round-robin-balance: true - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.7 - max-mamba-cache-size: 1024 - max-running-requests: 1024 - cuda-graph-max-bs: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "480" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-dep4-dep16-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-dep4-dep16-stp.yaml deleted file mode 100644 index 105e39bb2d..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-dep4-dep16-stp.yaml +++ /dev/null @@ -1,171 +0,0 @@ -schema: 2 -name: "qwen3.5-4p1d-dep4-dep16" - - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - wheel: "1.5.0.dev20260917" -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 3 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 4 - workers: 4 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - mem-fraction-static: 0.8 - chunked-prefill-size: 65536 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: "follow_bootstrap_room" - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.80 - max-mamba-cache-size: 2048 - max-running-requests: 2048 - cuda-graph-max-bs-decode: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "1024" - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml deleted file mode 100644 index e0cc1ab0a3..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml +++ /dev/null @@ -1,186 +0,0 @@ -# Qwen3.5 FP8 GB300 disaggregated MTP 4P1D DEP4/DEP16 configuration. - -schema: 2 -name: "qwen3.5-fp8-gb300-mtp-8k1k-4p1d-dep4-dep16" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 3 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 4 - workers: 4 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - mem-fraction-static: 0.7 - chunked-prefill-size: 98304 - max-prefill-tokens: 24576 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - moe-dense-tp-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - prefill-round-robin-balance: true - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.7 - max-mamba-cache-size: 1024 - max-running-requests: 1024 - cuda-graph-max-bs: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "768" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml deleted file mode 100644 index 510e6146a5..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml +++ /dev/null @@ -1,186 +0,0 @@ -# Qwen3.5 FP8 GB300 disaggregated MTP 6P1D DEP4/DEP16 configuration. - -schema: 2 -name: "qwen3.5-fp8-gb300-mtp-8k1k-6p1d-dep4-dep16" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 5 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 6 - workers: 6 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31001 - mem-fraction-static: 0.7 - chunked-prefill-size: 65536 - max-prefill-tokens: 16384 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - moe-dense-tp-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - prefill-round-robin-balance: true - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31001 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.7 - max-mamba-cache-size: 1024 - max-running-requests: 1024 - cuda-graph-max-bs: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "1280" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml deleted file mode 100644 index 0759135190..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml +++ /dev/null @@ -1,186 +0,0 @@ -# Qwen3.5 FP8 GB300 disaggregated MTP 7P1D DEP4/DEP16 configuration. - -schema: 2 -name: "qwen3.5-fp8-gb300-mtp-8k1k-7p1d-dep4-dep16" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 6 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 7 - workers: 7 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31001 - mem-fraction-static: 0.7 - chunked-prefill-size: 65536 - max-prefill-tokens: 16384 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - moe-dense-tp-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - prefill-round-robin-balance: true - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31001 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.7 - max-mamba-cache-size: 2048 - max-running-requests: 2048 - cuda-graph-max-bs: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "1344" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-dep4-dep16-stp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-dep4-dep16-stp.yaml deleted file mode 100644 index 2f5e4e196c..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-dep4-dep16-stp.yaml +++ /dev/null @@ -1,167 +0,0 @@ -schema: 2 -name: "qwen3.5-8p1d-dep4-dep16" - - -sbatch_directives: - mem: "0" - -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated -dynamo: - install: true - - source: - wheel: "1.5.0.dev20260917" -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 4 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 8 - workers: 8 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - mem-fraction-static: 0.8 - chunked-prefill-size: 65536 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - enable-dp-attention: true - enable-dp-lm-head: true - load-balance-method: "follow_bootstrap_room" - mamba-radix-cache-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31000 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.80 - max-mamba-cache-size: 2048 - max-running-requests: 2048 - cuda-graph-max-bs-decode: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - random_range_ratio: 0.8 - concurrencies: "2048x4096" diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml deleted file mode 100644 index 505b8cef0e..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml +++ /dev/null @@ -1,186 +0,0 @@ -# Qwen3.5 FP8 GB300 disaggregated MTP 8P1D DEP4/DEP16 configuration. - -schema: 2 -name: "qwen3.5-fp8-gb300-mtp-8k1k-8p1d-dep4-dep16" - -sbatch_directives: - mem: "0" - -dynamo: - install: true - - source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b -frontend: - type: dynamo - enable_multiple_frontends: true - num_additional_frontends: 7 - nginx_container: nginx - -model: - path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" - precision: "fp8" - -resources: - gpu_type: "gb300" - gpus_per_node: 4 -engine: sglang -roles: - prefill: - nodes: 8 - workers: 8 - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - SGLANG_DG_CACHE_DIR: "/configs/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - trust-remote-code: true - - tensor-parallel-size: 4 - data-parallel-size: 4 - expert-parallel-size: 4 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 2048 - mamba-ssm-dtype: "bfloat16" - disaggregation-mode: "prefill" - disable-radix-cache: true - disaggregation-bootstrap-port: 31001 - mem-fraction-static: 0.7 - chunked-prefill-size: 98304 - max-prefill-tokens: 24576 - load-balance-method: "round_robin" - watchdog-timeout: 1000000 - disable-cuda-graph: true - log-level: "info" - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "flashinfer_trtllm" - - decode: - nodes: 4 - workers: 1 - - env: - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0" - SGLANG_ENABLE_SPEC_V2: "1" - NO_COLOR: "1" - TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "3600" - TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: "3600" - TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: "3600" - PYTHONUNBUFFERED: "1" - NCCL_MNNVL_ENABLE: "1" - NCCL_CUMEM_ENABLE: "1" - MC_FORCE_MNNVL: "1" - MC_TE_METRIC: "true" - SGLANG_DG_CACHE_DIR: "/tmp/deepgemm-cache" - FLASHINFER_WORKSPACE_BASE: "/configs/flashinfer-cache" - SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000" - SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000" - SGLANG_DECODE_BOOTSTRAP_TIMEOUT: "1000" - SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: "1" - SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" - SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: "0" - SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: "1" - SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: "512" - SGLANG_HEALTH_CHECK_TIMEOUT: "1800" - SGLANG_HEALTH_STARTING_OK: "1" - SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: "0" - - args: - served-model-name: "Qwen/Qwen3.5-397B-A17B-FP8" - model-path: "/model/" - random-seed: 42 - trust-remote-code: true - quantization: "fp8" - kv-cache-dtype: "fp8_e4m3" - - tensor-parallel-size: 16 - data-parallel-size: 16 - expert-parallel-size: 16 - moe-dense-tp-size: 1 - enable-dp-attention: true - enable-dp-lm-head: true - prefill-round-robin-balance: true - - mamba-scheduler-strategy: "no_buffer" - mamba-track-interval: 128 - mamba-ssm-dtype: "bfloat16" - - speculative-algorithm: "EAGLE" - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - - disaggregation-mode: "decode" - disable-radix-cache: true - disaggregation-bootstrap-port: 31001 - - chunked-prefill-size: 4096 - context-length: 16384 - mem-fraction-static: 0.7 - max-mamba-cache-size: 2048 - max-running-requests: 2048 - cuda-graph-max-bs: 128 - watchdog-timeout: 1000000 - - page-size: 64 - attention-backend: "trtllm_mha" - moe-runner-backend: "deep_gemm" - moe-a2a-backend: "deepep" - deepep-mode: "low_latency" - ep-dispatch-algorithm: "static" - eplb-algorithm: "deepseek" - - decode-log-interval: 1 - stream-interval: 50 - -benchmark: - type: "sa-bench" - isl: 8192 - osl: 1024 - req_rate: "inf" - num_prompts_mult: 20 - num_warmup_mult: 2 - random_range_ratio: 0.8 - concurrencies: "1920x2304" - use_chat_template: true - -telemetry: - enabled: true - collect_interval_ms: 1000 - storage_subdir: power - required: true - startup_timeout_seconds: 120 - request_timeout_seconds: 2 - dcgm_exporter: - container_image: dcgm-exporter - port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml new file mode 100644 index 0000000000..da1c1729cd --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml @@ -0,0 +1,929 @@ +# srt-slurm recipes for qwen3.5/sglang/gb300-fp8/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_. + +schema: 2 + +base: + sbatch_directives: + mem: '0' + dynamo: + install: true + source: {} + frontend: + type: dynamo + enable_multiple_frontends: true + nginx_container: nginx + model: + path: qwen3.5-fp8 + precision: fp8 + resources: + gpu_type: gb300 + gpus_per_node: 4 + engine: sglang + roles: + prefill: + env: + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '3600' + TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: '3600' + TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: '3600' + PYTHONUNBUFFERED: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + MC_FORCE_MNNVL: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + args: + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + model-path: /model/ + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + trust-remote-code: true + attention-backend: trtllm_mha + mamba-ssm-dtype: bfloat16 + moe-runner-backend: flashinfer_trtllm + disable-radix-cache: true + disaggregation-mode: prefill + decode: + workers: 1 + env: + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '0' + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: '3600' + TORCH_NCCL_WATCHDOG_TIMEOUT_SEC: '3600' + TORCH_NCCL_HEARTBEAT_TIMEOUT_SEC: '3600' + PYTHONUNBUFFERED: '1' + NCCL_MNNVL_ENABLE: '1' + NCCL_CUMEM_ENABLE: '1' + MC_FORCE_MNNVL: '1' + MC_TE_METRIC: 'true' + FLASHINFER_WORKSPACE_BASE: /configs/flashinfer-cache + SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: '100000' + SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: '100000' + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: '100000' + SGLANG_DECODE_BOOTSTRAP_TIMEOUT: '1000' + SGLANG_HACK_SEQ_BOOTSTRAP_ROOM: '1' + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: 'True' + SGLANG_USE_MESSAGE_QUEUE_BROADCASTER: '0' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + model-path: /model/ + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + trust-remote-code: true + attention-backend: trtllm_mha + mamba-ssm-dtype: bfloat16 + disable-radix-cache: true + stream-interval: 50 + disaggregation-mode: decode + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + random_range_ratio: 0.8 + +# Qwen3.5 FP8 GB300 disaggregated MTP 1P1D TP4/TP4 configuration. +override_disagg_1p1d_p_tp4_d_tp4_b1024_c1x2x8_mtp: + name: qwen3.5-fp8-gb300-mtp-8k1k-1p1d-tp4-tp4 + dynamo: + source: + rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + frontend: + num_additional_frontends: 1 + model: + container: lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 + roles: + prefill: + nodes: 1 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 2048 + max-running-requests: 1024 + mem-fraction-static: 0.8 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + context-length: 16384 + cuda-graph-max-bs: 1024 + decode-log-interval: 1 + stream-interval: 50 + decode: + nodes: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + args: + tensor-parallel-size: 4 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 1024 + moe-runner-backend: flashinfer_trtllm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 1024 + mem-fraction-static: 0.8 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + context-length: 16384 + cuda-graph-max-bs: 1024 + decode-log-interval: 1 + benchmark: + concurrencies: 1x2x8 + use_chat_template: true + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 19401 + +# Qwen3.5 FP8 GB300 disaggregated MTP 1P1D TEP8/TEP8 configuration. +override_disagg_1p1d_p_tp8_ep8_d_tp8_ep8_b1024_c32x48x80_mtp: + name: qwen3.5-fp8-gb300-mtp-8k1k-1p1d-tep8-tep8 + dynamo: + source: + rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + frontend: + num_additional_frontends: 1 + model: + container: lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 + roles: + prefill: + nodes: 2 + workers: 1 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 8 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 2048 + max-running-requests: 1024 + mem-fraction-static: 0.8 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + context-length: 9236 + cuda-graph-max-bs: 320 + decode-log-interval: 50 + stream-interval: 50 + random-seed: 42 + data-parallel-size: 1 + expert-parallel-size: 8 + max-total-tokens: 128000 + scheduler-recv-interval: 10 + disaggregation-transfer-backend: mooncake + decode: + nodes: 2 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + args: + tensor-parallel-size: 8 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 1024 + moe-runner-backend: flashinfer_trtllm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 1024 + mem-fraction-static: 0.8 + chunked-prefill-size: 4096 + max-prefill-tokens: 16384 + context-length: 9236 + cuda-graph-max-bs: 320 + decode-log-interval: 50 + random-seed: 42 + data-parallel-size: 1 + expert-parallel-size: 8 + max-total-tokens: 2200000 + scheduler-recv-interval: 10 + disaggregation-transfer-backend: mooncake + benchmark: + concurrencies: 32x48x80 + use_chat_template: true + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 19401 + +override_disagg_1p1d_tp4_tp4_stp: + name: qwen3.5-1p1d-tp4-tp4 + dynamo: + source: + wheel: 1.5.0.dev20260917 + frontend: + num_additional_frontends: 1 + model: + container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 + roles: + prefill: + nodes: 1 + workers: 1 + args: + tensor-parallel-size: 4 + max-running-requests: 1024 + mem-fraction-static: 0.8 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + context-length: 16384 + decode-log-interval: 1 + stream-interval: 50 + cuda-graph-max-bs-decode: 1024 + decode: + nodes: 1 + env: + SGLANG_DG_CACHE_DIR: /configs/deepgemm-cache + SGLANG_HEALTH_CHECK_TIMEOUT: '3600' + args: + tensor-parallel-size: 4 + moe-runner-backend: flashinfer_trtllm + max-running-requests: 1024 + mem-fraction-static: 0.8 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + context-length: 16384 + decode-log-interval: 1 + cuda-graph-max-bs-decode: 1024 + benchmark: + concurrencies: 1x2x4x8x16x32x64x128 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + # 9401 is already bound by the cluster-level exporter on im-gb300 nodes; + # use a port outside that range. + port: 19401 + +# Qwen3.5 FP8 GB300 disaggregated MTP 3P1D DEP4/DEP16 configuration. +override_disagg_3p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c480_mtp: + name: qwen3.5-fp8-gb300-mtp-8k1k-3p1d-dep4-dep16 + dynamo: + source: + rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + frontend: + num_additional_frontends: 2 + model: + container: lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 + roles: + prefill: + nodes: 3 + workers: 3 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 65536 + max-prefill-tokens: 16384 + random-seed: 42 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + decode: + nodes: 4 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 1024 + moe-runner-backend: deep_gemm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 1024 + mem-fraction-static: 0.7 + chunked-prefill-size: 4096 + context-length: 16384 + cuda-graph-max-bs: 128 + decode-log-interval: 1 + random-seed: 42 + data-parallel-size: 16 + expert-parallel-size: 16 + moe-dense-tp-size: 1 + enable-dp-attention: true + enable-dp-lm-head: true + prefill-round-robin-balance: true + disaggregation-bootstrap-port: 31000 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + benchmark: + concurrencies: '480' + use_chat_template: true + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 19401 + +override_disagg_4p1d_dep4_dep16_stp: + name: qwen3.5-4p1d-dep4-dep16 + dynamo: + source: + wheel: 1.5.0.dev20260917 + frontend: + num_additional_frontends: 3 + model: + container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 + roles: + prefill: + nodes: 4 + workers: 4 + env: + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-track-interval: 2048 + mem-fraction-static: 0.8 + chunked-prefill-size: 65536 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + mamba-radix-cache-strategy: no_buffer + decode: + nodes: 4 + env: + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-track-interval: 128 + max-mamba-cache-size: 2048 + moe-runner-backend: deep_gemm + max-running-requests: 2048 + mem-fraction-static: 0.8 + chunked-prefill-size: 4096 + context-length: 16384 + decode-log-interval: 1 + data-parallel-size: 16 + expert-parallel-size: 16 + cuda-graph-max-bs-decode: 128 + enable-dp-attention: true + enable-dp-lm-head: true + disaggregation-bootstrap-port: 31000 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + load-balance-method: follow_bootstrap_room + mamba-radix-cache-strategy: no_buffer + benchmark: + concurrencies: '1024' + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 19401 + +# Qwen3.5 FP8 GB300 disaggregated MTP 4P1D DEP4/DEP16 configuration. +override_disagg_4p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c768_mtp: + name: qwen3.5-fp8-gb300-mtp-8k1k-4p1d-dep4-dep16 + dynamo: + source: + rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + frontend: + num_additional_frontends: 3 + model: + container: lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 + roles: + prefill: + nodes: 4 + workers: 4 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 98304 + max-prefill-tokens: 24576 + random-seed: 42 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + decode: + nodes: 4 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 1024 + moe-runner-backend: deep_gemm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 1024 + mem-fraction-static: 0.7 + chunked-prefill-size: 4096 + context-length: 16384 + cuda-graph-max-bs: 128 + decode-log-interval: 1 + random-seed: 42 + data-parallel-size: 16 + expert-parallel-size: 16 + moe-dense-tp-size: 1 + enable-dp-attention: true + enable-dp-lm-head: true + prefill-round-robin-balance: true + disaggregation-bootstrap-port: 31000 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + benchmark: + concurrencies: '768' + use_chat_template: true + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 19401 + +# Qwen3.5 FP8 GB300 disaggregated MTP 6P1D DEP4/DEP16 configuration. +override_disagg_6p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c1280_mtp: + name: qwen3.5-fp8-gb300-mtp-8k1k-6p1d-dep4-dep16 + dynamo: + source: + rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + frontend: + num_additional_frontends: 5 + model: + container: lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 + roles: + prefill: + nodes: 6 + workers: 6 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 65536 + max-prefill-tokens: 16384 + random-seed: 42 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + disaggregation-bootstrap-port: 31001 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + decode: + nodes: 4 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 1024 + moe-runner-backend: deep_gemm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 1024 + mem-fraction-static: 0.7 + chunked-prefill-size: 4096 + context-length: 16384 + cuda-graph-max-bs: 128 + decode-log-interval: 1 + random-seed: 42 + data-parallel-size: 16 + expert-parallel-size: 16 + moe-dense-tp-size: 1 + enable-dp-attention: true + enable-dp-lm-head: true + prefill-round-robin-balance: true + disaggregation-bootstrap-port: 31001 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + benchmark: + concurrencies: '1280' + use_chat_template: true + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 19401 + +# Qwen3.5 FP8 GB300 disaggregated MTP 7P1D DEP4/DEP16 configuration. +override_disagg_7p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b2048_c1344_mtp: + name: qwen3.5-fp8-gb300-mtp-8k1k-7p1d-dep4-dep16 + dynamo: + source: + rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + frontend: + num_additional_frontends: 6 + model: + container: lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 + roles: + prefill: + nodes: 7 + workers: 7 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 65536 + max-prefill-tokens: 16384 + random-seed: 42 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + disaggregation-bootstrap-port: 31001 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + decode: + nodes: 4 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 2048 + moe-runner-backend: deep_gemm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 4096 + context-length: 16384 + cuda-graph-max-bs: 128 + decode-log-interval: 1 + random-seed: 42 + data-parallel-size: 16 + expert-parallel-size: 16 + moe-dense-tp-size: 1 + enable-dp-attention: true + enable-dp-lm-head: true + prefill-round-robin-balance: true + disaggregation-bootstrap-port: 31001 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + benchmark: + concurrencies: '1344' + use_chat_template: true + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 19401 + +override_disagg_8p1d_dep4_dep16_stp: + name: qwen3.5-8p1d-dep4-dep16 + dynamo: + source: + wheel: 1.5.0.dev20260917 + frontend: + num_additional_frontends: 4 + model: + container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 + roles: + prefill: + nodes: 8 + workers: 8 + env: + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-track-interval: 2048 + mem-fraction-static: 0.8 + chunked-prefill-size: 65536 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + disaggregation-bootstrap-port: 31000 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + mamba-radix-cache-strategy: no_buffer + decode: + nodes: 4 + env: + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-track-interval: 128 + max-mamba-cache-size: 2048 + moe-runner-backend: deep_gemm + max-running-requests: 2048 + mem-fraction-static: 0.8 + chunked-prefill-size: 4096 + context-length: 16384 + decode-log-interval: 1 + data-parallel-size: 16 + expert-parallel-size: 16 + cuda-graph-max-bs-decode: 128 + enable-dp-attention: true + enable-dp-lm-head: true + disaggregation-bootstrap-port: 31000 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + load-balance-method: follow_bootstrap_room + mamba-radix-cache-strategy: no_buffer + benchmark: + concurrencies: 2048x4096 + services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + +# Qwen3.5 FP8 GB300 disaggregated MTP 8P1D DEP4/DEP16 configuration. +override_disagg_8p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b2048_c1920x2304_mtp: + name: qwen3.5-fp8-gb300-mtp-8k1k-8p1d-dep4-dep16 + dynamo: + source: + rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + frontend: + num_additional_frontends: 7 + model: + container: lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 + roles: + prefill: + nodes: 8 + workers: 8 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_STARTING_OK: '1' + SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: '0' + args: + tensor-parallel-size: 4 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 98304 + max-prefill-tokens: 24576 + random-seed: 42 + data-parallel-size: 4 + expert-parallel-size: 4 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + disaggregation-bootstrap-port: 31001 + load-balance-method: round_robin + watchdog-timeout: 1000000 + disable-cuda-graph: true + log-level: info + page-size: 64 + decode: + nodes: 4 + env: + SGLANG_ENABLE_SPEC_V2: '1' + NO_COLOR: '1' + SGLANG_DG_CACHE_DIR: /tmp/deepgemm-cache + SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' + SGLANG_HEALTH_CHECK_TIMEOUT: '1800' + SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' + args: + tensor-parallel-size: 16 + mamba-scheduler-strategy: no_buffer + mamba-track-interval: 128 + max-mamba-cache-size: 2048 + moe-runner-backend: deep_gemm + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + max-running-requests: 2048 + mem-fraction-static: 0.7 + chunked-prefill-size: 4096 + context-length: 16384 + cuda-graph-max-bs: 128 + decode-log-interval: 1 + random-seed: 42 + data-parallel-size: 16 + expert-parallel-size: 16 + moe-dense-tp-size: 1 + enable-dp-attention: true + enable-dp-lm-head: true + prefill-round-robin-balance: true + disaggregation-bootstrap-port: 31001 + watchdog-timeout: 1000000 + page-size: 64 + moe-a2a-backend: deepep + deepep-mode: low_latency + ep-dispatch-algorithm: static + eplb-algorithm: deepseek + benchmark: + concurrencies: 1920x2304 + use_chat_template: true + num_prompts_mult: 20 + num_warmup_mult: 2 + telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 19401 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp0-c2150.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp0-c2150.yaml deleted file mode 100644 index da222ed0ff..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp0-c2150.yaml +++ /dev/null @@ -1,166 +0,0 @@ -schema: 2 -name: disagg-gb300-10p1d-dep1-dep8-c2150-stp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 3 - workers: 10 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 2 - workers: 1 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - - 136 - - 144 - - 152 - - 160 - - 168 - - 176 - - 184 - - 192 - - 200 - - 208 - - 216 - - 224 - - 232 - - 240 - - 248 - - 256 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 256 - max_num_tokens: 256 - max_seq_len: 9472 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 2150 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-11p1d-dep16-b64-eplb0-mtp0-c1076.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-11p1d-dep16-b64-eplb0-mtp0-c1076.yaml deleted file mode 100644 index e8ae211d16..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-11p1d-dep16-b64-eplb0-mtp0-c1076.yaml +++ /dev/null @@ -1,144 +0,0 @@ -schema: 2 -name: disagg-gb300-11p1d-dep1-dep16-c1076-stp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 3 - workers: 11 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - transceiver_runtime: PYTHON - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 4 - workers: 1 - gpus: 16 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - max_batch_size: 64 - max_num_tokens: 64 - max_seq_len: 9472 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 16 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 1076 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-11p1d-dep8-b128-eplb0-mtp3-c1229.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-11p1d-dep8-b128-eplb0-mtp3-c1229.yaml deleted file mode 100644 index e3c25c88ac..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-11p1d-dep8-b128-eplb0-mtp3-c1229.yaml +++ /dev/null @@ -1,173 +0,0 @@ -schema: 2 -name: disagg-gb300-11p1d-dep1-dep8-c1229-mtp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 3 - workers: 11 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - transceiver_runtime: PYTHON - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 2 - workers: 1 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 12 - - 16 - - 20 - - 24 - - 28 - - 32 - - 36 - - 40 - - 44 - - 48 - - 52 - - 56 - - 60 - - 64 - - 68 - - 72 - - 76 - - 80 - - 84 - - 88 - - 92 - - 96 - - 100 - - 104 - - 108 - - 112 - - 116 - - 120 - - 124 - - 128 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 128 - max_num_tokens: 512 - max_seq_len: 9472 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 1229 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-16p1d-dep16-b128-eplb0-mtp0-c2253.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-16p1d-dep16-b128-eplb0-mtp0-c2253.yaml deleted file mode 100644 index b804ea9c6f..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-16p1d-dep16-b128-eplb0-mtp0-c2253.yaml +++ /dev/null @@ -1,152 +0,0 @@ -schema: 2 -name: disagg-gb300-16p1d-dep1-dep16-c2253-stp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 4 - workers: 16 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - transceiver_runtime: PYTHON - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 4 - workers: 1 - gpus: 16 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - max_batch_size: 128 - max_num_tokens: 128 - max_seq_len: 9472 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 16 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 2253 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-17p2d-dep8-b64-eplb0-mtp3-c1126.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-17p2d-dep8-b64-eplb0-mtp3-c1126.yaml deleted file mode 100644 index 9b9c31e8d4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-17p2d-dep8-b64-eplb0-mtp3-c1126.yaml +++ /dev/null @@ -1,155 +0,0 @@ -schema: 2 -name: disagg-gb300-17p2d-dep1-dep8-c1126-mtp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 5 - workers: 17 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 4 - workers: 2 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 12 - - 16 - - 20 - - 24 - - 28 - - 32 - - 36 - - 40 - - 44 - - 48 - - 52 - - 56 - - 60 - - 64 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.8 - max_batch_size: 64 - max_num_tokens: 256 - max_seq_len: 9472 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 1126 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p2d-tep8-b16-eplb0-mtp0-c42.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p2d-tep8-b16-eplb0-mtp0-c42.yaml deleted file mode 100644 index 5bf05178f4..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p2d-tep8-b16-eplb0-mtp0-c42.yaml +++ /dev/null @@ -1,132 +0,0 @@ -schema: 2 -name: disagg-gb300-1p2d-dep1-tep8-c42-stp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - max_batch_size: 32 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 4 - workers: 2 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 16 - max_num_tokens: 16 - max_seq_len: 9472 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 42 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p2d-tep8-b8-eplb0-mtp3-c20.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p2d-tep8-b8-eplb0-mtp3-c20.yaml deleted file mode 100644 index bcc1ba4916..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p2d-tep8-b8-eplb0-mtp3-c20.yaml +++ /dev/null @@ -1,137 +0,0 @@ -schema: 2 -name: disagg-gb300-1p2d-dep1-tep8-c20-mtp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - max_batch_size: 32 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 4 - workers: 2 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 8 - max_num_tokens: 32 - max_seq_len: 9472 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 20 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0-c8.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0-c8.yaml deleted file mode 100644 index 25d2d802d3..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0-c8.yaml +++ /dev/null @@ -1,128 +0,0 @@ -schema: 2 -name: disagg-gb300-1p4d-dep2-tep8-c8-stp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 2 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - max_batch_size: 32 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 2 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 2 - trust_remote_code: true - decode: - nodes: 8 - workers: 4 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 1 - max_num_tokens: 1 - max_seq_len: 9472 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 8 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3-c12.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3-c12.yaml deleted file mode 100644 index 1f6e44e916..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3-c12.yaml +++ /dev/null @@ -1,136 +0,0 @@ -schema: 2 -name: disagg-gb300-1p4d-dep1-tep8-c12-mtp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - max_batch_size: 32 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 8 - workers: 4 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 1 - max_num_tokens: 4 - max_seq_len: 9472 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 12 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b2-eplb0-mtp3-c8.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b2-eplb0-mtp3-c8.yaml deleted file mode 100644 index 40d27d0507..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b2-eplb0-mtp3-c8.yaml +++ /dev/null @@ -1,138 +0,0 @@ -schema: 2 -name: disagg-gb300-1p4d-dep1-tep8-c8-mtp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - max_batch_size: 32 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 8 - workers: 4 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 2 - max_num_tokens: 8 - max_seq_len: 9472 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 8 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp0-c24.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp0-c24.yaml deleted file mode 100644 index 0c6d5b3cc7..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp0-c24.yaml +++ /dev/null @@ -1,130 +0,0 @@ -schema: 2 -name: disagg-gb300-1p4d-dep1-tep8-c24-stp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: trtllm -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - max_batch_size: 32 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 8 - workers: 4 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 4 - max_num_tokens: 4 - max_seq_len: 9472 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 24 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-24p1d-dep16-b512-eplb0-mtp0-c8192-mtp-sweep.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-24p1d-dep16-b512-eplb0-mtp0-c8192-mtp-sweep.yaml deleted file mode 100644 index 410e07db36..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-24p1d-dep16-b512-eplb0-mtp0-c8192-mtp-sweep.yaml +++ /dev/null @@ -1,198 +0,0 @@ -schema: 2 -name: disagg-gb300-24p1d-dep1-dep16-c8192-stp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 6 - workers: 24 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 4 - workers: 1 - gpus: 16 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - - 136 - - 144 - - 152 - - 160 - - 168 - - 176 - - 184 - - 192 - - 200 - - 208 - - 216 - - 224 - - 232 - - 240 - - 248 - - 256 - - 264 - - 272 - - 280 - - 288 - - 296 - - 304 - - 312 - - 320 - - 328 - - 336 - - 344 - - 352 - - 360 - - 368 - - 376 - - 384 - - 392 - - 400 - - 408 - - 416 - - 424 - - 432 - - 440 - - 448 - - 456 - - 464 - - 472 - - 480 - - 488 - - 496 - - 504 - - 512 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - max_batch_size: 512 - max_num_tokens: 512 - max_seq_len: 9472 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 16 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 8192 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-24p1d-dep16-b512-eplb0-mtp0-c8192-stp-sweep.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-24p1d-dep16-b512-eplb0-mtp0-c8192-stp-sweep.yaml deleted file mode 100644 index 410e07db36..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-24p1d-dep16-b512-eplb0-mtp0-c8192-stp-sweep.yaml +++ /dev/null @@ -1,198 +0,0 @@ -schema: 2 -name: disagg-gb300-24p1d-dep1-dep16-c8192-stp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 6 - workers: 24 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 4 - workers: 1 - gpus: 16 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - - 136 - - 144 - - 152 - - 160 - - 168 - - 176 - - 184 - - 192 - - 200 - - 208 - - 216 - - 224 - - 232 - - 240 - - 248 - - 256 - - 264 - - 272 - - 280 - - 288 - - 296 - - 304 - - 312 - - 320 - - 328 - - 336 - - 344 - - 352 - - 360 - - 368 - - 376 - - 384 - - 392 - - 400 - - 408 - - 416 - - 424 - - 432 - - 440 - - 448 - - 456 - - 464 - - 472 - - 480 - - 488 - - 496 - - 504 - - 512 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - max_batch_size: 512 - max_num_tokens: 512 - max_seq_len: 9472 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 16 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 8192 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-26p1d-dep16-b256-eplb0-mtp2-c4301.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-26p1d-dep16-b256-eplb0-mtp2-c4301.yaml deleted file mode 100644 index f16e546399..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-26p1d-dep16-b256-eplb0-mtp2-c4301.yaml +++ /dev/null @@ -1,205 +0,0 @@ -schema: 2 -name: disagg-gb300-26p1d-dep1-dep16-c4301-mtp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 7 - workers: 26 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - transceiver_runtime: PYTHON - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 2 - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 4 - workers: 1 - gpus: 16 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 12 - - 16 - - 20 - - 24 - - 28 - - 32 - - 36 - - 40 - - 44 - - 48 - - 52 - - 56 - - 60 - - 64 - - 68 - - 72 - - 76 - - 80 - - 84 - - 88 - - 92 - - 96 - - 100 - - 104 - - 108 - - 112 - - 116 - - 120 - - 124 - - 128 - - 132 - - 136 - - 140 - - 144 - - 148 - - 152 - - 156 - - 160 - - 164 - - 168 - - 172 - - 176 - - 180 - - 184 - - 188 - - 192 - - 196 - - 200 - - 204 - - 208 - - 212 - - 216 - - 220 - - 224 - - 228 - - 232 - - 236 - - 240 - - 244 - - 248 - - 252 - - 256 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - max_batch_size: 256 - max_num_tokens: 768 - max_seq_len: 9472 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 2 - stream_interval: 100 - tensor_parallel_size: 16 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 4301 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p1d-dep16-b16-eplb0-mtp0-c282.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p1d-dep16-b16-eplb0-mtp0-c282.yaml deleted file mode 100644 index 04390fc340..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p1d-dep16-b16-eplb0-mtp0-c282.yaml +++ /dev/null @@ -1,136 +0,0 @@ -schema: 2 -name: disagg-gb300-4p1d-dep1-dep16-c282-stp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 1 - workers: 4 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 4 - workers: 1 - gpus: 16 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - max_batch_size: 16 - max_num_tokens: 16 - max_seq_len: 9472 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 16 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 282 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p3d-tep8-b32-eplb0-mtp3-c126.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p3d-tep8-b32-eplb0-mtp3-c126.yaml deleted file mode 100644 index 5e35f10696..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p3d-tep8-b32-eplb0-mtp3-c126.yaml +++ /dev/null @@ -1,145 +0,0 @@ -schema: 2 -name: disagg-gb300-4p3d-dep1-tep8-c126-mtp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 1 - workers: 4 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - max_batch_size: 32 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 6 - workers: 3 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 12 - - 16 - - 20 - - 24 - - 28 - - 32 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 9472 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 126 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p3d-tep8-b64-eplb0-mtp0-c210.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p3d-tep8-b64-eplb0-mtp0-c210.yaml deleted file mode 100644 index 4cc6f60a02..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p3d-tep8-b64-eplb0-mtp0-c210.yaml +++ /dev/null @@ -1,142 +0,0 @@ -schema: 2 -name: disagg-gb300-4p3d-dep1-tep8-c210-stp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 1 - workers: 4 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - max_batch_size: 32 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 6 - workers: 3 - gpus: 8 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - enable_padding: true - enable_attention_dp: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.9 - max_batch_size: 64 - max_num_tokens: 64 - max_seq_len: 9472 - moe_config: - backend: TRTLLM - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 210 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-5p1d-dep16-b8-eplb0-mtp3-c154.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-5p1d-dep16-b8-eplb0-mtp3-c154.yaml deleted file mode 100644 index 8bc5d4e327..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-5p1d-dep16-b8-eplb0-mtp3-c154.yaml +++ /dev/null @@ -1,143 +0,0 @@ -schema: 2 -name: disagg-gb300-5p1d-dep1-dep16-c154-mtp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 2 - workers: 5 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - transceiver_runtime: PYTHON - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 4 - workers: 1 - gpus: 16 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - max_batch_size: 8 - max_num_tokens: 32 - max_seq_len: 9472 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 100 - tensor_parallel_size: 16 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 154 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep16-b32-eplb0-mtp0-c563.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep16-b32-eplb0-mtp0-c563.yaml deleted file mode 100644 index 3f430da1ce..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep16-b32-eplb0-mtp0-c563.yaml +++ /dev/null @@ -1,138 +0,0 @@ -schema: 2 -name: disagg-gb300-7p1d-dep1-dep16-c563-stp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 2 - workers: 7 - gpus: 1 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - CTX_LOAD_STAGGER_S: '180' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.7 - max_batch_size: 2 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - tensor_parallel_size: 1 - trust_remote_code: true - decode: - nodes: 4 - workers: 1 - gpus: 16 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: false - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - max_batch_size: 32 - max_num_tokens: 32 - max_seq_len: 9472 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - stream_interval: 100 - tensor_parallel_size: 16 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 563 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep16-b32-eplb0-mtp3-c666.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep16-b32-eplb0-mtp3-c666.yaml deleted file mode 100644 index ade5667c13..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep16-b32-eplb0-mtp3-c666.yaml +++ /dev/null @@ -1,143 +0,0 @@ -schema: 2 -name: disagg-gb300-7p1d-dep2-dep16-c666-mtp -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - wheel: 1.4.0.dev20260807 - -identity: - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - sequential_node_start: 2 -roles: - prefill: - nodes: 4 - workers: 7 - gpus: 2 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - disable_overlap_scheduler: true - enable_attention_dp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.85 - max_batch_size: 32 - max_num_tokens: 16896 - max_seq_len: 8448 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 2 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - tensor_parallel_size: 2 - trust_remote_code: true - decode: - nodes: 4 - workers: 1 - gpus: 16 - env: - TLLM_LOG_LEVEL: INFO - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRTLLM_ENABLE_PDL: '1' - NCCL_GRAPH_MIXING_SUPPORT: '0' - MIMALLOC_PURGE_DELAY: '0' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - args: - cache_transceiver_config: - backend: UCX - kv_transfer_timeout_ms: 600000 - max_tokens_in_buffer: 16384 - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 12 - - 16 - - 20 - - 24 - - 28 - - 32 - enable_padding: true - enable_attention_dp: true - enable_lm_head_tp_in_adp: true - kv_cache_config: - dtype: fp8 - enable_block_reuse: false - free_gpu_memory_fraction: 0.75 - max_batch_size: 32 - max_num_tokens: 128 - max_seq_len: 9472 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - speculative_config: - decoding_type: MTP - max_draft_len: 3 - stream_interval: 100 - tensor_parallel_size: 16 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: true - env: - ETCD_LEASE_TTL: '120' - DYN_TOKENIZER: "fastokens" - -benchmark: - type: sa-bench - isl: 8192 - osl: 1024 - concurrencies: - - 666 - req_rate: inf - num_prompts_mult: 16 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml new file mode 100644 index 0000000000..c03f59981c --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml @@ -0,0 +1,1087 @@ +# srt-slurm recipes for qwen3.5/trtllm/gb300-fp4/8k1k: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_. + +schema: 2 + +base: + model: + path: qwen3.5-fp4 + container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 + precision: fp4 + dynamo: + install: true + source: + wheel: 1.4.0.dev20260807 + identity: + container: + image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 + slurm: + time_limit: 04:00:00 + health_check: + max_attempts: 540 + interval_seconds: 10 + resources: + gpu_type: gb300 + gpus_per_node: 4 + roles: + prefill: + env: + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + TRTLLM_ENABLE_PDL: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + MIMALLOC_PURGE_DELAY: '0' + UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp + args: + cache_transceiver_config: + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + cuda_graph_config: null + disable_overlap_scheduler: true + enable_attention_dp: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + max_num_tokens: 16896 + max_seq_len: 8448 + moe_config: + backend: CUTEDSL + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + pipeline_parallel_size: 1 + print_iter_log: true + trust_remote_code: true + decode: + env: + TLLM_LOG_LEVEL: INFO + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + TRTLLM_ENABLE_PDL: '1' + NCCL_GRAPH_MIXING_SUPPORT: '0' + MIMALLOC_PURGE_DELAY: '0' + UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp + args: + cache_transceiver_config: + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 16384 + cuda_graph_config: + enable_padding: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + max_seq_len: 9472 + moe_config: + use_low_precision_moe_combine: true + num_postprocess_workers: 4 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + pipeline_parallel_size: 1 + print_iter_log: true + stream_interval: 100 + trust_remote_code: true + frontend: + type: dynamo + env: + ETCD_LEASE_TTL: '120' + DYN_TOKENIZER: fastokens + benchmark: + type: sa-bench + isl: 8192 + osl: 1024 + req_rate: inf + num_prompts_mult: 16 + +override_disagg_10p1d_dep8_b256_eplb0_mtp0_c2150: + name: disagg-gb300-10p1d-dep1-dep8-c2150-stp + identity: + frameworks: + tensorrt_llm: 1.3.0rc24 + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 3 + workers: 10 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: NIXL + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 2 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + decode: + nodes: 2 + workers: 1 + gpus: 8 + args: + cache_transceiver_config: + backend: NIXL + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 256 + max_num_tokens: 256 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + frontend: + enable_multiple_frontends: true + benchmark: + concurrencies: [2150] + +override_disagg_11p1d_dep16_b64_eplb0_mtp0_c1076: + name: disagg-gb300-11p1d-dep1-dep16-c1076-stp + identity: + frameworks: + tensorrt_llm: 1.3.0rc24 + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 3 + workers: 11 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: NIXL + transceiver_runtime: PYTHON + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 2 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cache_transceiver_config: + backend: NIXL + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_batch_size: 64 + max_num_tokens: 64 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + frontend: + enable_multiple_frontends: true + benchmark: + concurrencies: [1076] + +override_disagg_11p1d_dep8_b128_eplb0_mtp3_c1229: + name: disagg-gb300-11p1d-dep1-dep8-c1229-mtp + identity: + frameworks: + tensorrt_llm: 1.3.0rc24 + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 3 + workers: 11 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: NIXL + transceiver_runtime: PYTHON + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 2 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + decode: + nodes: 2 + workers: 1 + gpus: 8 + args: + cache_transceiver_config: + backend: NIXL + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 12, 16, 20, 24, 28, 32, 36, 40, 44, 48, 52, 56, 60, 64, 68, 72, 76, 80, 84, 88, 92, 96, 100, 104, 108, 112, 116, 120, 124, 128] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 128 + max_num_tokens: 512 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + frontend: + enable_multiple_frontends: true + benchmark: + concurrencies: [1229] + +override_disagg_16p1d_dep16_b128_eplb0_mtp0_c2253: + name: disagg-gb300-16p1d-dep1-dep16-c2253-stp + identity: + frameworks: + tensorrt_llm: 1.3.0rc24 + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 4 + workers: 16 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: NIXL + transceiver_runtime: PYTHON + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 2 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cache_transceiver_config: + backend: NIXL + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_batch_size: 128 + max_num_tokens: 128 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + frontend: + enable_multiple_frontends: true + benchmark: + concurrencies: [2253] + +override_disagg_17p2d_dep8_b64_eplb0_mtp3_c1126: + name: disagg-gb300-17p2d-dep1-dep8-c1126-mtp + identity: + frameworks: + tensorrt_llm: 1.3.0rc24 + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 5 + workers: 17 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: NIXL + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 2 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + decode: + nodes: 4 + workers: 2 + gpus: 8 + args: + cache_transceiver_config: + backend: NIXL + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 12, 16, 20, 24, 28, 32, 36, 40, 44, 48, 52, 56, 60, 64] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 64 + max_num_tokens: 256 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + frontend: + enable_multiple_frontends: true + benchmark: + concurrencies: [1126] + +override_disagg_1p2d_tep8_b16_eplb0_mtp0_c42: + name: disagg-gb300-1p2d-dep1-tep8-c42-stp + engine: trtllm + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: UCX + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 32 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + decode: + nodes: 4 + workers: 2 + gpus: 8 + args: + cache_transceiver_config: + backend: UCX + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 16 + max_num_tokens: 16 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + frontend: + enable_multiple_frontends: false + benchmark: + concurrencies: [42] + +override_disagg_1p2d_tep8_b8_eplb0_mtp3_c20: + name: disagg-gb300-1p2d-dep1-tep8-c20-mtp + engine: trtllm + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: UCX + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 32 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + decode: + nodes: 4 + workers: 2 + gpus: 8 + args: + cache_transceiver_config: + backend: UCX + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 8 + max_num_tokens: 32 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + frontend: + enable_multiple_frontends: false + benchmark: + concurrencies: [20] + +override_disagg_1p4d_tep8_b1_eplb0_mtp0_c8: + name: disagg-gb300-1p4d-dep2-tep8-c8-stp + engine: trtllm + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 2 + args: + cache_transceiver_config: + backend: UCX + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 32 + moe_expert_parallel_size: 2 + tensor_parallel_size: 2 + decode: + nodes: 8 + workers: 4 + gpus: 8 + args: + cache_transceiver_config: + backend: UCX + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 1 + max_num_tokens: 1 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + frontend: + enable_multiple_frontends: false + benchmark: + concurrencies: [8] + +override_disagg_1p4d_tep8_b1_eplb0_mtp3_c12: + name: disagg-gb300-1p4d-dep1-tep8-c12-mtp + engine: trtllm + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: UCX + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 32 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + decode: + nodes: 8 + workers: 4 + gpus: 8 + args: + cache_transceiver_config: + backend: UCX + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 1 + max_num_tokens: 4 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + frontend: + enable_multiple_frontends: false + benchmark: + concurrencies: [12] + +override_disagg_1p4d_tep8_b2_eplb0_mtp3_c8: + name: disagg-gb300-1p4d-dep1-tep8-c8-mtp + identity: + frameworks: + tensorrt_llm: 1.3.0rc24 + engine: trtllm + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: UCX + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 32 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + decode: + nodes: 8 + workers: 4 + gpus: 8 + args: + cache_transceiver_config: + backend: UCX + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 2 + max_num_tokens: 8 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + frontend: + enable_multiple_frontends: false + benchmark: + concurrencies: [8] + +override_disagg_1p4d_tep8_b4_eplb0_mtp0_c24: + name: disagg-gb300-1p4d-dep1-tep8-c24-stp + engine: trtllm + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: UCX + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 32 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + decode: + nodes: 8 + workers: 4 + gpus: 8 + args: + cache_transceiver_config: + backend: UCX + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 4 + max_num_tokens: 4 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + frontend: + enable_multiple_frontends: false + benchmark: + concurrencies: [24] + +override_disagg_24p1d_dep16_b512_eplb0_mtp0_c8192_mtp_sweep: + name: disagg-gb300-24p1d-dep1-dep16-c8192-stp + identity: + frameworks: + tensorrt_llm: 1.3.0rc24 + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 6 + workers: 24 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: NIXL + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 2 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cache_transceiver_config: + backend: NIXL + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 264, 272, 280, 288, 296, 304, 312, 320, 328, 336, 344, 352, 360, 368, 376, 384, 392, 400, 408, 416, 424, 432, 440, 448, 456, 464, 472, 480, 488, 496, 504, 512] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_batch_size: 512 + max_num_tokens: 512 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + frontend: + enable_multiple_frontends: true + benchmark: + concurrencies: [8192] + +override_disagg_24p1d_dep16_b512_eplb0_mtp0_c8192_stp_sweep: + name: disagg-gb300-24p1d-dep1-dep16-c8192-stp + identity: + frameworks: + tensorrt_llm: 1.3.0rc24 + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 6 + workers: 24 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: NIXL + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 2 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cache_transceiver_config: + backend: NIXL + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 264, 272, 280, 288, 296, 304, 312, 320, 328, 336, 344, 352, 360, 368, 376, 384, 392, 400, 408, 416, 424, 432, 440, 448, 456, 464, 472, 480, 488, 496, 504, 512] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_batch_size: 512 + max_num_tokens: 512 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + frontend: + enable_multiple_frontends: true + benchmark: + concurrencies: [8192] + +override_disagg_26p1d_dep16_b256_eplb0_mtp2_c4301: + name: disagg-gb300-26p1d-dep1-dep16-c4301-mtp + identity: + frameworks: + tensorrt_llm: 1.3.0rc24 + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 7 + workers: 26 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: NIXL + transceiver_runtime: PYTHON + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 2 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + speculative_config: + decoding_type: MTP + max_draft_len: 2 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cache_transceiver_config: + backend: NIXL + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 12, 16, 20, 24, 28, 32, 36, 40, 44, 48, 52, 56, 60, 64, 68, 72, 76, 80, 84, 88, 92, 96, 100, 104, 108, 112, 116, 120, 124, 128, 132, 136, 140, 144, 148, 152, 156, 160, 164, 168, 172, 176, 180, 184, 188, 192, 196, 200, 204, 208, 212, 216, 220, 224, 228, 232, 236, 240, 244, 248, 252, 256] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_batch_size: 256 + max_num_tokens: 768 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + speculative_config: + decoding_type: MTP + max_draft_len: 2 + frontend: + enable_multiple_frontends: true + benchmark: + concurrencies: [4301] + +override_disagg_4p1d_dep16_b16_eplb0_mtp0_c282: + name: disagg-gb300-4p1d-dep1-dep16-c282-stp + identity: + frameworks: + tensorrt_llm: 1.3.0rc24 + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 1 + workers: 4 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: NIXL + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 2 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cache_transceiver_config: + backend: NIXL + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_batch_size: 16 + max_num_tokens: 16 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + frontend: + enable_multiple_frontends: false + benchmark: + concurrencies: [282] + +override_disagg_4p3d_tep8_b32_eplb0_mtp3_c126: + name: disagg-gb300-4p3d-dep1-tep8-c126-mtp + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 1 + workers: 4 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: UCX + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 32 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + decode: + nodes: 6 + workers: 3 + gpus: 8 + args: + cache_transceiver_config: + backend: UCX + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 12, 16, 20, 24, 28, 32] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 32 + max_num_tokens: 128 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + frontend: + enable_multiple_frontends: false + benchmark: + concurrencies: [126] + +override_disagg_4p3d_tep8_b64_eplb0_mtp0_c210: + name: disagg-gb300-4p3d-dep1-tep8-c210-stp + identity: + frameworks: + tensorrt_llm: 1.3.0rc24 + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 1 + workers: 4 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: UCX + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 32 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + decode: + nodes: 6 + workers: 3 + gpus: 8 + args: + cache_transceiver_config: + backend: UCX + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 64 + max_num_tokens: 64 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 8 + tensor_parallel_size: 8 + frontend: + enable_multiple_frontends: false + benchmark: + concurrencies: [210] + +override_disagg_5p1d_dep16_b8_eplb0_mtp3_c154: + name: disagg-gb300-5p1d-dep1-dep16-c154-mtp + identity: + frameworks: + tensorrt_llm: 1.3.0rc24 + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 2 + workers: 5 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: NIXL + transceiver_runtime: PYTHON + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 2 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cache_transceiver_config: + backend: NIXL + transceiver_runtime: PYTHON + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_batch_size: 8 + max_num_tokens: 32 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + frontend: + enable_multiple_frontends: false + benchmark: + concurrencies: [154] + +override_disagg_7p1d_dep16_b32_eplb0_mtp0_c563: + name: disagg-gb300-7p1d-dep1-dep16-c563-stp + identity: + frameworks: + tensorrt_llm: 1.3.0rc24 + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 2 + workers: 7 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + CTX_LOAD_STAGGER_S: '180' + args: + cache_transceiver_config: + backend: NIXL + kv_cache_config: + free_gpu_memory_fraction: 0.7 + max_batch_size: 2 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cache_transceiver_config: + backend: NIXL + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32] + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_batch_size: 32 + max_num_tokens: 32 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + frontend: + enable_multiple_frontends: true + benchmark: + concurrencies: [563] + +override_disagg_7p1d_dep16_b32_eplb0_mtp3_c666: + name: disagg-gb300-7p1d-dep2-dep16-c666-mtp + engine: + type: trtllm + sequential_node_start: 2 + roles: + prefill: + nodes: 4 + workers: 7 + gpus: 2 + args: + cache_transceiver_config: + backend: UCX + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 32 + moe_expert_parallel_size: 2 + tensor_parallel_size: 2 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cache_transceiver_config: + backend: UCX + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 12, 16, 20, 24, 28, 32] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_batch_size: 32 + max_num_tokens: 128 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 16 + tensor_parallel_size: 16 + speculative_config: + decoding_type: MTP + max_draft_len: 3 + frontend: + enable_multiple_frontends: true + benchmark: + concurrencies: [666] diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p1d-dep1-tep2-c44-b8-mtp-kvoffload.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p1d-dep1-tep2-c44-b8-mtp-kvoffload.yaml deleted file mode 100644 index f498c1d931..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p1d-dep1-tep2-c44-b8-mtp-kvoffload.yaml +++ /dev/null @@ -1,222 +0,0 @@ -schema: 2 -name: disagg-gb300-1p1d-dep1-tep2-c44-b8-mtp-kvoffload -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - -identity: - model: - repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 - -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 1 - env: - CUDA_SCALE_LAUNCH_QUEUES: 4x - MIMALLOC_ARENA_RESERVE: '0' - MIMALLOC_PURGE_DELAY: '' - NCCL_GRAPH_MIXING_SUPPORT: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TLLM_LOG_LEVEL: INFO - TRTLLM_ENABLE_PDL: '1' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - args: - attention_dp_config: - kv_cache_routing_conversation_affinity: true - cache_transceiver_config: - backend: NIXL - kv_transfer_sender_future_timeout_ms: 20 - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - enable_iter_perf_stats: true - enable_iter_req_stats: false - kv_cache_config: - block_reuse_config: - max_num_turns: 3 - policy: per_conversation - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - host_cache_size: 137438953472 - mamba_state_config: - additional_snapshot_offsets_from_end: - - 2 - periodic_snapshot_interval: 0 - pool_ratio: - - 0.8 - - 0.2 - tokens_per_block: 64 - use_kv_cache_manager_v2: true - max_batch_size: 64 - max_num_tokens: 8192 - max_seq_len: 262144 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - num_serve_frontends: 8 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - tensor_parallel_size: 1 - torch_compile_config: - capture_num_tokens: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 1024 - - 2048 - - 4096 - - 8192 - enable_fullgraph: true - enable_piecewise_cuda_graph: true - trust_remote_code: true - decode: - nodes: 1 - workers: 1 - gpus: 2 - env: - MIMALLOC_ARENA_RESERVE: '0' - MIMALLOC_PURGE_DELAY: '' - NCCL_GRAPH_MIXING_SUPPORT: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TLLM_LOG_LEVEL: INFO - TRTLLM_ENABLE_PDL: '1' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_sender_future_timeout_ms: 20 - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - enable_padding: true - enable_attention_dp: false - enable_iter_perf_stats: true - enable_iter_req_stats: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - avg_seq_len: 102150 - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.85 - host_cache_size: 137438953472 - tokens_per_block: 64 - use_kv_cache_manager_v2: true - max_batch_size: 8 - max_num_tokens: 56 - max_seq_len: 262148 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 2 - num_postprocess_workers: 4 - num_serve_frontends: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: GUARANTEED_NO_EVICT - speculative_config: - decoding_type: MTP - max_draft_len: 6 - stream_interval: 20 - tensor_parallel_size: 2 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: "120" - DYN_TOKENIZER_CACHE: "1" - DYN_TOKENIZER_CACHE_BYTES: "8000000000" - DYN_TOKENIZER: "fastokens" - args: - router-mode: kv - router-session-affinity-ttl-secs: '14400' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - SERVED_MODEL_NAME: Qwen3.5-397B-A17B-NVFP4-V2 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p7d-dep4-tep8-c7-b1-mtp-kvoffload.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p7d-dep4-tep8-c7-b1-mtp-kvoffload.yaml deleted file mode 100644 index 076f17ae91..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p7d-dep4-tep8-c7-b1-mtp-kvoffload.yaml +++ /dev/null @@ -1,219 +0,0 @@ -schema: 2 -name: disagg-gb300-1p7d-dep4-tep8-c7-b1-mtp-kvoffload -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - -identity: - model: - repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 - -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false -roles: - prefill: - nodes: 1 - workers: 1 - gpus: 4 - env: - CUDA_SCALE_LAUNCH_QUEUES: 4x - MIMALLOC_ARENA_RESERVE: '0' - MIMALLOC_PURGE_DELAY: '' - NCCL_GRAPH_MIXING_SUPPORT: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TLLM_LOG_LEVEL: INFO - TRTLLM_ENABLE_PDL: '1' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - args: - attention_dp_config: - kv_cache_routing_conversation_affinity: true - cache_transceiver_config: - backend: NIXL - kv_transfer_sender_future_timeout_ms: 20 - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - enable_iter_perf_stats: true - enable_iter_req_stats: false - kv_cache_config: - block_reuse_config: - max_num_turns: 3 - policy: per_conversation - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - host_cache_size: 137438953472 - mamba_state_config: - additional_snapshot_offsets_from_end: - - 2 - periodic_snapshot_interval: 0 - pool_ratio: - - 0.8 - - 0.2 - tokens_per_block: 64 - use_kv_cache_manager_v2: true - max_batch_size: 64 - max_num_tokens: 8192 - max_seq_len: 262144 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - num_serve_frontends: 8 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - tensor_parallel_size: 4 - torch_compile_config: - capture_num_tokens: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 1024 - - 2048 - - 4096 - - 8192 - enable_fullgraph: true - enable_piecewise_cuda_graph: true - trust_remote_code: true - decode: - nodes: 14 - workers: 7 - gpus: 8 - env: - MIMALLOC_ARENA_RESERVE: '0' - MIMALLOC_PURGE_DELAY: '' - NCCL_GRAPH_MIXING_SUPPORT: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TLLM_LOG_LEVEL: INFO - TRTLLM_ENABLE_PDL: '1' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_sender_future_timeout_ms: 20 - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_iter_perf_stats: true - enable_iter_req_stats: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - avg_seq_len: 102150 - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.85 - host_cache_size: 137438953472 - tokens_per_block: 64 - use_kv_cache_manager_v2: true - max_batch_size: 1 - max_num_tokens: 8 - max_seq_len: 262148 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - num_serve_frontends: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: GUARANTEED_NO_EVICT - speculative_config: - decoding_type: MTP - max_draft_len: 7 - stream_interval: 20 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: "120" - DYN_TOKENIZER_CACHE: "1" - DYN_TOKENIZER_CACHE_BYTES: "8000000000" - DYN_TOKENIZER: "fastokens" - args: - router-mode: kv - router-session-affinity-ttl-secs: '14400' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - SERVED_MODEL_NAME: Qwen3.5-397B-A17B-NVFP4-V2 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-2p2d-dep1-tep2-c52-b4-mtp-kvoffload.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-2p2d-dep1-tep2-c52-b4-mtp-kvoffload.yaml deleted file mode 100644 index 091438994f..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-2p2d-dep1-tep2-c52-b4-mtp-kvoffload.yaml +++ /dev/null @@ -1,221 +0,0 @@ -schema: 2 -name: disagg-gb300-2p2d-dep1-tep2-c52-b4-mtp-kvoffload -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - -identity: - model: - repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 - -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false -roles: - prefill: - nodes: 1 - workers: 2 - gpus: 1 - env: - CUDA_SCALE_LAUNCH_QUEUES: 4x - MIMALLOC_ARENA_RESERVE: '0' - MIMALLOC_PURGE_DELAY: '' - NCCL_GRAPH_MIXING_SUPPORT: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TLLM_LOG_LEVEL: INFO - TRTLLM_ENABLE_PDL: '1' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - args: - attention_dp_config: - kv_cache_routing_conversation_affinity: true - cache_transceiver_config: - backend: NIXL - kv_transfer_sender_future_timeout_ms: 20 - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - enable_iter_perf_stats: true - enable_iter_req_stats: false - kv_cache_config: - block_reuse_config: - max_num_turns: 3 - policy: per_conversation - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - host_cache_size: 137438953472 - mamba_state_config: - additional_snapshot_offsets_from_end: - - 2 - periodic_snapshot_interval: 0 - pool_ratio: - - 0.8 - - 0.2 - tokens_per_block: 64 - use_kv_cache_manager_v2: true - max_batch_size: 64 - max_num_tokens: 8192 - max_seq_len: 262144 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 1 - num_serve_frontends: 8 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - tensor_parallel_size: 1 - torch_compile_config: - capture_num_tokens: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 1024 - - 2048 - - 4096 - - 8192 - enable_fullgraph: true - enable_piecewise_cuda_graph: true - trust_remote_code: true - decode: - nodes: 1 - workers: 2 - gpus: 2 - env: - MIMALLOC_ARENA_RESERVE: '0' - MIMALLOC_PURGE_DELAY: '' - NCCL_GRAPH_MIXING_SUPPORT: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TLLM_LOG_LEVEL: INFO - TRTLLM_ENABLE_PDL: '1' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_sender_future_timeout_ms: 20 - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - enable_padding: true - enable_attention_dp: false - enable_iter_perf_stats: true - enable_iter_req_stats: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - avg_seq_len: 102150 - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.85 - host_cache_size: 137438953472 - tokens_per_block: 64 - use_kv_cache_manager_v2: true - max_batch_size: 4 - max_num_tokens: 28 - max_seq_len: 262148 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 2 - num_postprocess_workers: 4 - num_serve_frontends: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: GUARANTEED_NO_EVICT - speculative_config: - decoding_type: MTP - max_draft_len: 6 - stream_interval: 20 - tensor_parallel_size: 2 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: "120" - DYN_TOKENIZER_CACHE: "1" - DYN_TOKENIZER_CACHE_BYTES: "8000000000" - DYN_TOKENIZER: "fastokens" - args: - router-mode: kv - router-session-affinity-ttl-secs: '14400' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - SERVED_MODEL_NAME: Qwen3.5-397B-A17B-NVFP4-V2 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-2p3d-tep2-tep8-c96-b128-mtp-kvoffload.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-2p3d-tep2-tep8-c96-b128-mtp-kvoffload.yaml deleted file mode 100644 index f3134246e6..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-2p3d-tep2-tep8-c96-b128-mtp-kvoffload.yaml +++ /dev/null @@ -1,236 +0,0 @@ -schema: 2 -name: disagg-gb300-2p3d-tep2-tep8-c96-b128-mtp-kvoffload -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - -identity: - model: - repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 - -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false -roles: - prefill: - nodes: 1 - workers: 2 - gpus: 2 - env: - CUDA_SCALE_LAUNCH_QUEUES: 4x - MIMALLOC_ARENA_RESERVE: '0' - MIMALLOC_PURGE_DELAY: '' - NCCL_GRAPH_MIXING_SUPPORT: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TLLM_LOG_LEVEL: INFO - TRTLLM_ENABLE_PDL: '1' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - args: - attention_dp_config: - kv_cache_routing_conversation_affinity: true - cache_transceiver_config: - backend: NIXL - kv_transfer_sender_future_timeout_ms: 20 - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - disable_overlap_scheduler: false - enable_attention_dp: false - enable_chunked_prefill: true - enable_iter_perf_stats: true - enable_iter_req_stats: false - kv_cache_config: - block_reuse_config: - max_num_turns: 3 - policy: per_conversation - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - host_cache_size: 137438953472 - mamba_state_config: - additional_snapshot_offsets_from_end: - - 2 - periodic_snapshot_interval: 0 - pool_ratio: - - 0.8 - - 0.2 - tokens_per_block: 64 - use_kv_cache_manager_v2: true - max_batch_size: 64 - max_num_tokens: 16384 - max_seq_len: 262144 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 2 - num_serve_frontends: 8 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - tensor_parallel_size: 2 - torch_compile_config: - capture_num_tokens: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 1024 - - 2048 - - 4096 - - 8192 - - 16384 - enable_fullgraph: true - enable_piecewise_cuda_graph: true - trust_remote_code: true - decode: - nodes: 6 - workers: 3 - gpus: 8 - env: - MIMALLOC_ARENA_RESERVE: '0' - MIMALLOC_PURGE_DELAY: '' - NCCL_GRAPH_MIXING_SUPPORT: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TLLM_LOG_LEVEL: INFO - TRTLLM_ENABLE_PDL: '1' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_sender_future_timeout_ms: 20 - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - - 40 - - 48 - - 56 - - 64 - - 72 - - 80 - - 88 - - 96 - - 104 - - 112 - - 120 - - 128 - enable_padding: true - enable_attention_dp: false - enable_iter_perf_stats: true - enable_iter_req_stats: false - enable_lm_head_tp_in_adp: false - kv_cache_config: - avg_seq_len: 102150 - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.85 - host_cache_size: 137438953472 - tokens_per_block: 64 - use_kv_cache_manager_v2: true - max_batch_size: 128 - max_num_tokens: 896 - max_seq_len: 262148 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 8 - num_postprocess_workers: 4 - num_serve_frontends: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: GUARANTEED_NO_EVICT - speculative_config: - decoding_type: MTP - max_draft_len: 6 - stream_interval: 20 - tensor_parallel_size: 8 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: "120" - DYN_TOKENIZER_CACHE: "1" - DYN_TOKENIZER_CACHE_BYTES: "8000000000" - DYN_TOKENIZER: "fastokens" - args: - router-mode: kv - router-session-affinity-ttl-secs: '14400' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - SERVED_MODEL_NAME: Qwen3.5-397B-A17B-NVFP4-V2 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-3p1d-dep4-dep16-c565-b8-mtp-kvoffload.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-3p1d-dep4-dep16-c565-b8-mtp-kvoffload.yaml deleted file mode 100644 index cddc0d8a81..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-3p1d-dep4-dep16-c565-b8-mtp-kvoffload.yaml +++ /dev/null @@ -1,220 +0,0 @@ -schema: 2 -name: disagg-gb300-3p1d-dep4-dep16-c565-b8-mtp-kvoffload -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - -identity: - model: - repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 - -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false -roles: - prefill: - nodes: 3 - workers: 3 - gpus: 4 - env: - CUDA_SCALE_LAUNCH_QUEUES: 4x - MIMALLOC_ARENA_RESERVE: '0' - MIMALLOC_PURGE_DELAY: '' - NCCL_GRAPH_MIXING_SUPPORT: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TLLM_LOG_LEVEL: INFO - TRTLLM_ENABLE_PDL: '1' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - args: - attention_dp_config: - kv_cache_routing_conversation_affinity: true - cache_transceiver_config: - backend: NIXL - kv_transfer_sender_future_timeout_ms: 20 - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - enable_iter_perf_stats: true - enable_iter_req_stats: false - kv_cache_config: - block_reuse_config: - max_num_turns: 3 - policy: per_conversation - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - host_cache_size: 137438953472 - mamba_state_config: - additional_snapshot_offsets_from_end: - - 2 - periodic_snapshot_interval: 0 - pool_ratio: - - 0.8 - - 0.2 - tokens_per_block: 64 - use_kv_cache_manager_v2: true - max_batch_size: 64 - max_num_tokens: 8192 - max_seq_len: 262144 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - num_serve_frontends: 8 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - tensor_parallel_size: 4 - torch_compile_config: - capture_num_tokens: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 1024 - - 2048 - - 4096 - - 8192 - enable_fullgraph: true - enable_piecewise_cuda_graph: true - trust_remote_code: true - decode: - nodes: 4 - workers: 1 - gpus: 16 - env: - MIMALLOC_ARENA_RESERVE: '0' - MIMALLOC_PURGE_DELAY: '' - NCCL_GRAPH_MIXING_SUPPORT: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TLLM_LOG_LEVEL: INFO - TRTLLM_ENABLE_PDL: '1' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_sender_future_timeout_ms: 20 - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - enable_padding: true - enable_attention_dp: true - enable_iter_perf_stats: true - enable_iter_req_stats: false - enable_lm_head_tp_in_adp: true - kv_cache_config: - avg_seq_len: 102150 - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.85 - host_cache_size: 137438953472 - tokens_per_block: 64 - use_kv_cache_manager_v2: true - max_batch_size: 8 - max_num_tokens: 56 - max_seq_len: 262148 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 16 - num_postprocess_workers: 4 - num_serve_frontends: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: GUARANTEED_NO_EVICT - speculative_config: - decoding_type: MTP - max_draft_len: 6 - stream_interval: 20 - tensor_parallel_size: 16 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: "120" - DYN_TOKENIZER_CACHE: "1" - DYN_TOKENIZER_CACHE_BYTES: "8000000000" - DYN_TOKENIZER: "fastokens" - args: - router-mode: kv - router-session-affinity-ttl-secs: '14400' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - SERVED_MODEL_NAME: Qwen3.5-397B-A17B-NVFP4-V2 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-3p2d-dep4-dep4-c704-b32-mtp-kvoffload.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-3p2d-dep4-dep4-c704-b32-mtp-kvoffload.yaml deleted file mode 100644 index 3c0b23c7b6..0000000000 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-3p2d-dep4-dep4-c704-b32-mtp-kvoffload.yaml +++ /dev/null @@ -1,223 +0,0 @@ -schema: 2 -name: disagg-gb300-3p2d-dep4-dep4-c704-b32-mtp-kvoffload -model: - path: qwen3.5-fp4 - container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - precision: fp4 - -dynamo: - install: true - source: - git: https://github.com/cquil11/dynamo.git - rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 - -identity: - model: - repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - container: - image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 - frameworks: - tensorrt_llm: 1.3.0rc24 - -slurm: - time_limit: 04:00:00 -health_check: - max_attempts: 540 - interval_seconds: 10 -resources: - gpu_type: gb300 - gpus_per_node: 4 -engine: - type: trtllm - publish_events_and_metrics: false -roles: - prefill: - nodes: 3 - workers: 3 - gpus: 4 - env: - CUDA_SCALE_LAUNCH_QUEUES: 4x - MIMALLOC_ARENA_RESERVE: '0' - MIMALLOC_PURGE_DELAY: '' - NCCL_GRAPH_MIXING_SUPPORT: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TLLM_LOG_LEVEL: INFO - TRTLLM_ENABLE_PDL: '1' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - args: - attention_dp_config: - kv_cache_routing_conversation_affinity: true - cache_transceiver_config: - backend: NIXL - kv_transfer_sender_future_timeout_ms: 20 - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - disable_overlap_scheduler: false - enable_attention_dp: true - enable_chunked_prefill: true - enable_iter_perf_stats: true - enable_iter_req_stats: false - kv_cache_config: - block_reuse_config: - max_num_turns: 3 - policy: per_conversation - dtype: fp8 - enable_block_reuse: true - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - host_cache_size: 137438953472 - mamba_state_config: - additional_snapshot_offsets_from_end: - - 2 - periodic_snapshot_interval: 0 - pool_ratio: - - 0.8 - - 0.2 - tokens_per_block: 64 - use_kv_cache_manager_v2: true - max_batch_size: 64 - max_num_tokens: 8192 - max_seq_len: 262144 - moe_config: - backend: CUTEDSL - moe_expert_parallel_size: 4 - num_serve_frontends: 8 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - tensor_parallel_size: 4 - torch_compile_config: - capture_num_tokens: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 - - 256 - - 512 - - 1024 - - 2048 - - 4096 - - 8192 - enable_fullgraph: true - enable_piecewise_cuda_graph: true - trust_remote_code: true - decode: - nodes: 2 - workers: 2 - gpus: 4 - env: - MIMALLOC_ARENA_RESERVE: '0' - MIMALLOC_PURGE_DELAY: '' - NCCL_GRAPH_MIXING_SUPPORT: '0' - PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True - TLLM_LOG_LEVEL: INFO - TRTLLM_ENABLE_PDL: '1' - TRTLLM_PINNED_WEIGHT_STAGING: '1' - TRTLLM_SERVER_DISABLE_GC: '1' - TRTLLM_WORKER_DISABLE_GC: '1' - UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp - DYN_ENGINE_CONV_AFFINITY: "1" - DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - args: - cache_transceiver_config: - backend: NIXL - kv_transfer_sender_future_timeout_ms: 20 - kv_transfer_timeout_ms: 600000 - transceiver_runtime: PYTHON - cuda_graph_config: - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 24 - - 32 - enable_padding: true - enable_attention_dp: true - enable_iter_perf_stats: true - enable_iter_req_stats: false - enable_lm_head_tp_in_adp: true - kv_cache_config: - avg_seq_len: 102150 - dtype: fp8 - enable_block_reuse: false - event_buffer_max_size: 0 - free_gpu_memory_fraction: 0.8 - host_cache_size: 137438953472 - tokens_per_block: 64 - use_kv_cache_manager_v2: true - max_batch_size: 32 - max_num_tokens: 224 - max_seq_len: 262148 - moe_config: - backend: CUTEDSL - use_low_precision_moe_combine: true - moe_expert_parallel_size: 4 - num_postprocess_workers: 4 - num_serve_frontends: 4 - nvfp4_gemm_config: - allowed_backends: - - cutlass - - cublaslt - - cutedsl - - cuda_core - pipeline_parallel_size: 1 - print_iter_log: true - return_perf_metrics: false - scheduler_config: - capacity_scheduler_policy: GUARANTEED_NO_EVICT - speculative_config: - decoding_type: MTP - max_draft_len: 6 - stream_interval: 20 - tensor_parallel_size: 4 - trust_remote_code: true -frontend: - type: dynamo - enable_multiple_frontends: false - env: - ETCD_LEASE_TTL: "120" - DYN_TOKENIZER_CACHE: "1" - DYN_TOKENIZER_CACHE_BYTES: "8000000000" - DYN_TOKENIZER: "fastokens" - args: - router-mode: kv - router-session-affinity-ttl-secs: '14400' - active-decode-blocks-threshold: None - active-prefill-tokens-threshold: None - active-prefill-tokens-threshold-frac: None - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: '8000' - IS_MULTINODE: 'true' - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization - AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache - HF_HUB_CACHE: /hf_hub_cache - WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k - SERVED_MODEL_NAME: Qwen3.5-397B-A17B-NVFP4-V2 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml new file mode 100644 index 0000000000..d74c5b232d --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml @@ -0,0 +1,383 @@ +# srt-slurm recipes for qwen3.5/trtllm/gb300-fp4/agentx: shared settings in base, one override per +# benchmark configuration. Select one with +# CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml:override_. + +schema: 2 + +base: + model: + path: qwen3.5-fp4 + container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 + precision: fp4 + dynamo: + install: true + source: + git: https://github.com/cquil11/dynamo.git + rev: 2cbbdc863c336c2cfffa2a3668e2a6661801b259 + identity: + model: + repo: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + container: + image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc24 + frameworks: + tensorrt_llm: 1.3.0rc24 + slurm: + time_limit: 04:00:00 + health_check: + max_attempts: 540 + interval_seconds: 10 + resources: + gpu_type: gb300 + gpus_per_node: 4 + engine: + type: trtllm + publish_events_and_metrics: false + roles: + prefill: + env: + CUDA_SCALE_LAUNCH_QUEUES: 4x + MIMALLOC_ARENA_RESERVE: '0' + MIMALLOC_PURGE_DELAY: '' + NCCL_GRAPH_MIXING_SUPPORT: '0' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + TLLM_LOG_LEVEL: INFO + TRTLLM_ENABLE_PDL: '1' + TRTLLM_PINNED_WEIGHT_STAGING: '1' + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp + DYN_ENGINE_CONV_AFFINITY: '1' + DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + args: + attention_dp_config: + kv_cache_routing_conversation_affinity: true + cache_transceiver_config: + backend: NIXL + kv_transfer_sender_future_timeout_ms: 20 + kv_transfer_timeout_ms: 600000 + transceiver_runtime: PYTHON + cuda_graph_config: null + disable_overlap_scheduler: false + enable_chunked_prefill: true + enable_iter_perf_stats: true + enable_iter_req_stats: false + kv_cache_config: + block_reuse_config: + max_num_turns: 3 + policy: per_conversation + dtype: fp8 + enable_block_reuse: true + event_buffer_max_size: 0 + free_gpu_memory_fraction: 0.8 + host_cache_size: 137438953472 + mamba_state_config: + additional_snapshot_offsets_from_end: + - 2 + periodic_snapshot_interval: 0 + pool_ratio: + - 0.8 + - 0.2 + tokens_per_block: 64 + use_kv_cache_manager_v2: true + max_batch_size: 64 + max_seq_len: 262144 + moe_config: + backend: CUTEDSL + num_serve_frontends: 8 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + pipeline_parallel_size: 1 + print_iter_log: true + return_perf_metrics: false + scheduler_config: + capacity_scheduler_policy: MAX_UTILIZATION + torch_compile_config: + enable_fullgraph: true + enable_piecewise_cuda_graph: true + trust_remote_code: true + decode: + env: + MIMALLOC_ARENA_RESERVE: '0' + MIMALLOC_PURGE_DELAY: '' + NCCL_GRAPH_MIXING_SUPPORT: '0' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + TLLM_LOG_LEVEL: INFO + TRTLLM_ENABLE_PDL: '1' + TRTLLM_PINNED_WEIGHT_STAGING: '1' + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + UCX_TLS: cuda_ipc,cuda_copy,sm,self,tcp + DYN_ENGINE_CONV_AFFINITY: '1' + DYN_TRTLLM_SERVED_MODEL_NAME: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + args: + cache_transceiver_config: + backend: NIXL + kv_transfer_sender_future_timeout_ms: 20 + kv_transfer_timeout_ms: 600000 + transceiver_runtime: PYTHON + cuda_graph_config: + enable_padding: true + enable_iter_perf_stats: true + enable_iter_req_stats: false + kv_cache_config: + avg_seq_len: 102150 + dtype: fp8 + enable_block_reuse: false + event_buffer_max_size: 0 + host_cache_size: 137438953472 + tokens_per_block: 64 + use_kv_cache_manager_v2: true + max_seq_len: 262148 + moe_config: + backend: CUTEDSL + use_low_precision_moe_combine: true + num_postprocess_workers: 4 + num_serve_frontends: 4 + nvfp4_gemm_config: + allowed_backends: + - cutlass + - cublaslt + - cutedsl + - cuda_core + pipeline_parallel_size: 1 + print_iter_log: true + return_perf_metrics: false + scheduler_config: + capacity_scheduler_policy: GUARANTEED_NO_EVICT + speculative_config: + decoding_type: MTP + stream_interval: 20 + trust_remote_code: true + frontend: + type: dynamo + enable_multiple_frontends: false + env: + ETCD_LEASE_TTL: '120' + DYN_TOKENIZER_CACHE: '1' + DYN_TOKENIZER_CACHE_BYTES: '8000000000' + DYN_TOKENIZER: fastokens + args: + router-mode: kv + router-session-affinity-ttl-secs: '14400' + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: '8000' + IS_MULTINODE: 'true' + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: trtllm_kv_cache_utilization + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126_256k + SERVED_MODEL_NAME: Qwen3.5-397B-A17B-NVFP4-V2 + +override_disagg_1p1d_dep1_tep2_c44_b8_mtp_kvoffload: + name: disagg-gb300-1p1d-dep1-tep2-c44-b8-mtp-kvoffload + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + args: + enable_attention_dp: true + max_num_tokens: 8192 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + torch_compile_config: + capture_num_tokens: [1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 1024, 2048, 4096, 8192] + decode: + nodes: 1 + workers: 1 + gpus: 2 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 8 + max_num_tokens: 56 + moe_expert_parallel_size: 2 + speculative_config: + max_draft_len: 6 + tensor_parallel_size: 2 + +override_disagg_1p7d_dep4_tep8_c7_b1_mtp_kvoffload: + name: disagg-gb300-1p7d-dep4-tep8-c7-b1-mtp-kvoffload + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + args: + enable_attention_dp: true + max_num_tokens: 8192 + moe_expert_parallel_size: 4 + tensor_parallel_size: 4 + torch_compile_config: + capture_num_tokens: [1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 1024, 2048, 4096, 8192] + decode: + nodes: 14 + workers: 7 + gpus: 8 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 1 + max_num_tokens: 8 + moe_expert_parallel_size: 8 + speculative_config: + max_draft_len: 7 + tensor_parallel_size: 8 + +override_disagg_2p2d_dep1_tep2_c52_b4_mtp_kvoffload: + name: disagg-gb300-2p2d-dep1-tep2-c52-b4-mtp-kvoffload + roles: + prefill: + nodes: 1 + workers: 2 + gpus: 1 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + args: + enable_attention_dp: true + max_num_tokens: 8192 + moe_expert_parallel_size: 1 + tensor_parallel_size: 1 + torch_compile_config: + capture_num_tokens: [1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 1024, 2048, 4096, 8192] + decode: + nodes: 1 + workers: 2 + gpus: 2 + env: + TRT_LLM_DISABLE_LOAD_WEIGHTS_IN_PARALLEL: '1' + args: + cuda_graph_config: + batch_sizes: [1, 2, 4] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 4 + max_num_tokens: 28 + moe_expert_parallel_size: 2 + speculative_config: + max_draft_len: 6 + tensor_parallel_size: 2 + +override_disagg_2p3d_tep2_tep8_c96_b128_mtp_kvoffload: + name: disagg-gb300-2p3d-tep2-tep8-c96-b128-mtp-kvoffload + roles: + prefill: + nodes: 1 + workers: 2 + gpus: 2 + args: + enable_attention_dp: false + max_num_tokens: 16384 + moe_expert_parallel_size: 2 + tensor_parallel_size: 2 + torch_compile_config: + capture_num_tokens: [1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 1024, 2048, 4096, 8192, 16384] + decode: + nodes: 6 + workers: 3 + gpus: 8 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128] + enable_attention_dp: false + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 128 + max_num_tokens: 896 + moe_expert_parallel_size: 8 + speculative_config: + max_draft_len: 6 + tensor_parallel_size: 8 + +override_disagg_3p1d_dep4_dep16_c565_b8_mtp_kvoffload: + name: disagg-gb300-3p1d-dep4-dep16-c565-b8-mtp-kvoffload + roles: + prefill: + nodes: 3 + workers: 3 + gpus: 4 + args: + enable_attention_dp: true + max_num_tokens: 8192 + moe_expert_parallel_size: 4 + tensor_parallel_size: 4 + torch_compile_config: + capture_num_tokens: [1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 1024, 2048, 4096, 8192] + decode: + nodes: 4 + workers: 1 + gpus: 16 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.85 + max_batch_size: 8 + max_num_tokens: 56 + moe_expert_parallel_size: 16 + speculative_config: + max_draft_len: 6 + tensor_parallel_size: 16 + +override_disagg_3p2d_dep4_dep4_c704_b32_mtp_kvoffload: + name: disagg-gb300-3p2d-dep4-dep4-c704-b32-mtp-kvoffload + roles: + prefill: + nodes: 3 + workers: 3 + gpus: 4 + args: + enable_attention_dp: true + max_num_tokens: 8192 + moe_expert_parallel_size: 4 + tensor_parallel_size: 4 + torch_compile_config: + capture_num_tokens: [1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 1024, 2048, 4096, 8192] + decode: + nodes: 2 + workers: 2 + gpus: 4 + args: + cuda_graph_config: + batch_sizes: [1, 2, 4, 8, 16, 24, 32] + enable_attention_dp: true + enable_lm_head_tp_in_adp: true + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 32 + max_num_tokens: 224 + moe_expert_parallel_size: 4 + speculative_config: + max_draft_len: 6 + tensor_parallel_size: 4 diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 7b195bc2fe..7b22bb38d2 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -23,7 +23,7 @@ dsr1-fp4-b200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b200-fp4/8k1k/mtp/ctx1_gen1_dep8_batch8_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p1d-dep8-b8-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_1p1d_dep8_b8_eplb0_mtp3" decode: num-worker: 1 tp: 8 @@ -38,7 +38,7 @@ dsr1-fp4-b200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b200-fp4/8k1k/mtp/ctx1_gen3_tep8_batch16_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_1p3d_tep8_b16_eplb0_mtp3" decode: num-worker: 3 tp: 8 @@ -53,7 +53,7 @@ dsr1-fp4-b200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b200-fp4/8k1k/mtp/ctx1_gen5_tep8_batch1_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b1-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_1p5d_tep8_b1_eplb0_mtp3" decode: num-worker: 5 tp: 8 @@ -68,7 +68,7 @@ dsr1-fp4-b200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b200-fp4/8k1k/mtp/ctx1_gen5_tep8_batch8_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b8-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_1p5d_tep8_b8_eplb0_mtp3" decode: num-worker: 5 tp: 8 @@ -83,7 +83,7 @@ dsr1-fp4-b200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b200-fp4/8k1k/mtp/ctx3_gen1_dep8_batch64_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_3p1d_dep8_b64_eplb0_mtp3" decode: num-worker: 1 tp: 8 @@ -98,7 +98,7 @@ dsr1-fp4-b200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b200-fp4/8k1k/mtp/ctx5_gen1_dep8_batch192_eplb0_mtp1.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-5p1d-dep8-b192-eplb0-mtp1.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_5p1d_dep8_b192_eplb0_mtp1" decode: num-worker: 1 tp: 8 @@ -113,7 +113,7 @@ dsr1-fp4-b200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b200-fp4/8k1k/mtp/ctx5_gen2_dep8_batch32_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-5p2d-dep8-b32-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_5p2d_dep8_b32_eplb0_mtp3" decode: num-worker: 2 tp: 8 @@ -129,7 +129,7 @@ dsr1-fp4-b200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b200-fp4/8k1k/stp/ctx1_gen5_tep8_batch1_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b1-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_1p5d_tep8_b1_eplb0_mtp0" decode: num-worker: 5 tp: 8 @@ -143,7 +143,7 @@ dsr1-fp4-b200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b200-fp4/8k1k/stp/ctx1_gen5_tep8_batch8_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-1p5d-tep8-b16-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_1p5d_tep8_b16_eplb0_mtp0" decode: num-worker: 5 tp: 8 @@ -157,7 +157,7 @@ dsr1-fp4-b200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b200-fp4/8k1k/stp/ctx2_gen5_tep8_batch64_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-2p5d-tep8-b64-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_2p5d_tep8_b64_eplb0_mtp0" decode: num-worker: 5 tp: 8 @@ -171,7 +171,7 @@ dsr1-fp4-b200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b200-fp4/8k1k/stp/ctx4_gen1_dep8_batch192_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-4p1d-dep8-b192-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_4p1d_dep8_b192_eplb0_mtp0" decode: num-worker: 1 tp: 8 @@ -185,7 +185,7 @@ dsr1-fp4-b200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b200-fp4/8k1k/stp/ctx4_gen3_dep8_batch32_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-4p3d-dep8-b32-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_4p3d_dep8_b32_eplb0_mtp0" decode: num-worker: 3 tp: 8 @@ -199,7 +199,7 @@ dsr1-fp4-b200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b200-fp4/8k1k/stp/ctx7_gen2_dep8_batch128_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/disagg-7p2d-dep8-b128-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b200-fp4/8k1k/variants.yaml:override_disagg_7p2d_dep8_b128_eplb0_mtp0" decode: num-worker: 2 tp: 8 @@ -470,7 +470,7 @@ dsr1-fp4-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp4/8k1k/mtp/ctx10_gen1_dep8_batch256_eplb0_mtp1.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp1.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_10p1d_dep8_b256_eplb0_mtp1" decode: num-worker: 1 tp: 8 @@ -485,7 +485,7 @@ dsr1-fp4-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp4/8k1k/mtp/ctx1_gen4_tep4_batch8_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep4-b8-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep4_b8_eplb0_mtp3" decode: num-worker: 4 tp: 4 @@ -500,7 +500,7 @@ dsr1-fp4-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp4/8k1k/mtp/ctx1_gen4_tep8_batch1_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp3" decode: num-worker: 4 tp: 8 @@ -515,7 +515,7 @@ dsr1-fp4-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp4/8k1k/mtp/ctx1_gen4_tep8_batch4_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b4_eplb0_mtp3" decode: num-worker: 4 tp: 8 @@ -530,7 +530,7 @@ dsr1-fp4-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp4/8k1k/mtp/ctx3_gen1_dep8_batch16_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-3p1d-dep8-b16-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_3p1d_dep8_b16_eplb0_mtp3" decode: num-worker: 1 tp: 8 @@ -545,7 +545,7 @@ dsr1-fp4-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp4/8k1k/mtp/ctx9_gen1_dep8_batch128_eplb0_mtp1.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-9p1d-dep8-b128-eplb0-mtp1.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_9p1d_dep8_b128_eplb0_mtp1" decode: num-worker: 1 tp: 8 @@ -561,7 +561,7 @@ dsr1-fp4-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp4/8k1k/stp/ctx1_gen3_tep4_batch32_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep4-b32-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p3d_tep4_b32_eplb0_mtp0" decode: num-worker: 3 tp: 4 @@ -575,7 +575,7 @@ dsr1-fp4-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp4/8k1k/stp/ctx1_gen3_tep8_batch16_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p3d_tep8_b16_eplb0_mtp0" decode: num-worker: 3 tp: 8 @@ -589,7 +589,7 @@ dsr1-fp4-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp4/8k1k/stp/ctx1_gen3_tep8_batch1_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p3d-tep8-b1-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p3d_tep8_b1_eplb0_mtp0" decode: num-worker: 3 tp: 8 @@ -603,7 +603,7 @@ dsr1-fp4-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp4/8k1k/stp/ctx1_gen4_tep4_batch2_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-1p4d-tep4-b2-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep4_b2_eplb0_mtp0" decode: num-worker: 4 tp: 4 @@ -617,7 +617,7 @@ dsr1-fp4-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp4/8k1k/stp/ctx5_gen2_dep8_batch32_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-5p2d-dep8-b32-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_5p2d_dep8_b32_eplb0_mtp0" decode: num-worker: 2 tp: 8 @@ -631,7 +631,7 @@ dsr1-fp4-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp4/8k1k/stp/ctx6_gen1_dep8_batch128_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-6p1d-dep8-b128-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_6p1d_dep8_b128_eplb0_mtp0" decode: num-worker: 1 tp: 8 @@ -645,7 +645,7 @@ dsr1-fp4-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp4/8k1k/stp/ctx8_gen1_dep8_batch256_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/disagg-8p1d-dep8-b256-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp4/8k1k/variants.yaml:override_disagg_8p1d_dep8_b256_eplb0_mtp0" decode: num-worker: 1 tp: 8 @@ -676,7 +676,7 @@ dsr1-fp8-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp8/8k1k/mtp/ctx1_gen2_tp8_batch16_eplb0_mtp3_40.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p2d-tp8-b16-eplb0-mtp3-c40.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_1p2d_tp8_b16_eplb0_mtp3_c40" decode: num-worker: 2 tp: 8 @@ -691,7 +691,7 @@ dsr1-fp8-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp8/8k1k/mtp/ctx1_gen4_tp8_batch1_eplb0_mtp3_8.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b1-eplb0-mtp3-c8.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tp8_b1_eplb0_mtp3_c8" decode: num-worker: 4 tp: 8 @@ -706,7 +706,7 @@ dsr1-fp8-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp8/8k1k/mtp/ctx1_gen4_tp8_batch4_eplb0_mtp3_20.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b4-eplb0-mtp3-c20.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tp8_b4_eplb0_mtp3_c20" decode: num-worker: 4 tp: 8 @@ -721,7 +721,7 @@ dsr1-fp8-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp8/8k1k/mtp/ctx1_gen1_dp8_batch8_eplb0_mtp3_72.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p1d-dp8-b8-eplb0-mtp3-c72.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_1p1d_dp8_b8_eplb0_mtp3_c72" decode: num-worker: 1 tp: 8 @@ -736,7 +736,7 @@ dsr1-fp8-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp8/8k1k/mtp/ctx2_gen1_dp8_batch16_eplb0_mtp3_144.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-2p1d-dp8-b16-eplb0-mtp3-c144.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_2p1d_dp8_b16_eplb0_mtp3_c144" decode: num-worker: 1 tp: 8 @@ -751,7 +751,7 @@ dsr1-fp8-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp8/8k1k/mtp/ctx4_gen1_dp8_batch64_eplb0_mtp2_512.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-4p1d-dp8-b64-eplb0-mtp2-c512.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_4p1d_dp8_b64_eplb0_mtp2_c512" decode: num-worker: 1 tp: 8 @@ -769,7 +769,7 @@ dsr1-fp8-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp8/8k1k/stp/ctx1_gen4_tp8_batch16_eplb0_mtp0_64.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p4d-tp8-b16-eplb0-mtp0-c64.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tp8_b16_eplb0_mtp0_c64" decode: num-worker: 4 tp: 8 @@ -783,7 +783,7 @@ dsr1-fp8-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp8/8k1k/stp/ctx1_gen8_tp8_batch2_eplb0_mtp0_16.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-1p8d-tp8-b1-eplb0-mtp0-c16.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_1p8d_tp8_b1_eplb0_mtp0_c16" decode: num-worker: 8 tp: 8 @@ -797,7 +797,7 @@ dsr1-fp8-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp8/8k1k/stp/ctx2_gen1_dp8_batch32_eplb0_mtp0_256.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-2p1d-dp8-b32-eplb0-mtp0-c256.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_2p1d_dp8_b32_eplb0_mtp0_c256" decode: num-worker: 1 tp: 8 @@ -811,7 +811,7 @@ dsr1-fp8-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp8/8k1k/stp/ctx3_gen1_dp8_batch64_eplb0_mtp0_512.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-3p1d-dp8-b64-eplb0-mtp0-c512.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_3p1d_dp8_b64_eplb0_mtp0_c512" decode: num-worker: 1 tp: 8 @@ -825,7 +825,7 @@ dsr1-fp8-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp8/8k1k/stp/ctx3_gen5_tp8_batch64_eplb0_mtp0_256.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-3p5d-tp8-b64-eplb0-mtp0-c256.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_3p5d_tp8_b64_eplb0_mtp0_c256" decode: num-worker: 5 tp: 8 @@ -839,7 +839,7 @@ dsr1-fp8-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp8/8k1k/stp/ctx5_gen1_dp8_batch128_eplb0_mtp0_1075.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-5p1d-dp8-b128-eplb0-mtp0-c1075.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_5p1d_dp8_b128_eplb0_mtp0_c1075" decode: num-worker: 1 tp: 8 @@ -853,7 +853,7 @@ dsr1-fp8-b300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/b300-fp8/8k1k/stp/ctx7_gen1_dep8_batch384_eplb0_mtp0_3072.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/disagg-7p1d-dep8-b384-eplb0-mtp0-c3072.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/b300-fp8/8k1k/variants.yaml:override_disagg_7p1d_dep8_b384_eplb0_mtp0_c3072" decode: num-worker: 1 tp: 8 @@ -1272,7 +1272,7 @@ kimik3-fp4-h200-vllm-agentic-latency: ep: 32 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/h200-fp4/agentx/agg-tp16dp2ep32-latency.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/h200-fp4/agentx/variants.yaml:override_agg_tp16dp2ep32_latency" kimik3-fp4-h200-vllm-agentic-balanced: image: vllm/vllm-openai:kimi-k3 @@ -1298,7 +1298,7 @@ kimik3-fp4-h200-vllm-agentic-balanced: ep: 32 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/h200-fp4/agentx/agg-tp8dp4ep32-balanced.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/h200-fp4/agentx/variants.yaml:override_agg_tp8dp4ep32_balanced" kimik3-fp4-h200-vllm-agentic-simple: image: vllm/vllm-openai:kimi-k3 @@ -1325,7 +1325,7 @@ kimik3-fp4-h200-vllm-agentic-simple: ep: 32 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/h200-fp4/agentx/agg-tp8dp4ep32-vllm-simple.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/h200-fp4/agentx/variants.yaml:override_agg_tp8dp4ep32_vllm_simple" # NOTE: At the time of submission, https://docs.vllm.ai/projects/recipes/en/latest/moonshotai/Kimi-K2.5.html # does not have a B300-specific recipe, so this config reuses the existing # Kimi-K2.5 FP4 B200 vLLM recipe as-is until B300-specific tuning is available. @@ -1598,7 +1598,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/mtp/c1_ctx1_gen7_tep8_batch1_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b1-eplb0-mtp3-c9.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p7d_tep8_b1_eplb0_mtp3_c9" decode: num-worker: 7 tp: 8 @@ -1613,7 +1613,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/mtp/c4_ctx1_gen7_tep8_batch32_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b32-eplb0-mtp3-c28.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p7d_tep8_b32_eplb0_mtp3_c28" decode: num-worker: 7 tp: 8 @@ -1628,7 +1628,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/mtp/c8_ctx1_gen6_tep8_batch32_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p6d-tep8-b32-eplb0-mtp3-c48.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p6d_tep8_b32_eplb0_mtp3_c48" decode: num-worker: 6 tp: 8 @@ -1643,7 +1643,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/mtp/c16_ctx1_gen3_tep8_batch32_eplb0_mtp2.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p3d-tep8-b32-eplb0-mtp2-c48.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b32_eplb0_mtp2_c48" decode: num-worker: 3 tp: 8 @@ -1658,7 +1658,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/mtp/c32_ctx3_gen5_tep8_batch32_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p5d-tep8-b32-eplb0-mtp3-c160.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_3p5d_tep8_b32_eplb0_mtp3_c160" decode: num-worker: 5 tp: 8 @@ -1673,7 +1673,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/mtp/c64_ctx1_gen1_dep8_batch32_eplb0_mtp2.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p1d-dep8-b32-eplb0-mtp2-c64.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p1d_dep8_b32_eplb0_mtp2_c64" decode: num-worker: 1 tp: 8 @@ -1688,7 +1688,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/mtp/c128_ctx2_gen1_dep8_batch32_eplb0_mtp2.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p1d-dep8-b32-eplb0-mtp2-c128.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_2p1d_dep8_b32_eplb0_mtp2_c128" decode: num-worker: 1 tp: 8 @@ -1703,7 +1703,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/mtp/c256_ctx3_gen1_dep8_batch32_eplb0_mtp2.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b32-eplb0-mtp2-c256.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_3p1d_dep8_b32_eplb0_mtp2_c256" decode: num-worker: 1 tp: 8 @@ -1718,7 +1718,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/mtp/c512_ctx3_gen1_dep8_batch64_eplb0_mtp1.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp1-c512.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_3p1d_dep8_b64_eplb0_mtp1_c512" decode: num-worker: 1 tp: 8 @@ -1733,7 +1733,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/stp/c1_ctx1_gen7_tep8_batch1_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b1-eplb0-mtp0-c9.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p7d_tep8_b1_eplb0_mtp0_c9" decode: num-worker: 7 tp: 8 @@ -1747,7 +1747,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/stp/c4_ctx1_gen7_tep8_batch32_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p7d-tep8-b32-eplb0-mtp0-c28.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p7d_tep8_b32_eplb0_mtp0_c28" decode: num-worker: 7 tp: 8 @@ -1761,7 +1761,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/stp/c8_ctx1_gen6_tep8_batch16_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p6d-tep8-b16-eplb0-mtp0-c48.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p6d_tep8_b16_eplb0_mtp0_c48" decode: num-worker: 6 tp: 8 @@ -1775,7 +1775,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/stp/c16_ctx1_gen3_tep8_batch32_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p3d-tep8-b32-eplb0-mtp0-c48.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b32_eplb0_mtp0_c48" decode: num-worker: 3 tp: 8 @@ -1789,7 +1789,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/stp/c32_ctx2_gen5_tep8_batch128_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p5d-tep8-b128-eplb0-mtp0-c160.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_2p5d_tep8_b128_eplb0_mtp0_c160" decode: num-worker: 5 tp: 8 @@ -1803,7 +1803,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/stp/c64_ctx2_gen3_dep8_batch128_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-2p3d-dep8-b128-eplb0-mtp0-c192.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_2p3d_dep8_b128_eplb0_mtp0_c192" decode: num-worker: 3 tp: 8 @@ -1817,7 +1817,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/stp/c128_ctx1_gen1_dep8_batch256_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-1p1d-dep8-b256-eplb0-mtp0-c128.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_1p1d_dep8_b256_eplb0_mtp0_c128" decode: num-worker: 1 tp: 8 @@ -1831,7 +1831,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/stp/c256_ctx5_gen3_dep8_batch256_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-5p3d-dep8-b256-eplb0-mtp0-c768.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_5p3d_dep8_b256_eplb0_mtp0_c768" decode: num-worker: 3 tp: 8 @@ -1845,7 +1845,7 @@ dsr1-fp8-h200-dynamo-trt: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h200/8k1k/stp/c512_ctx3_gen1_dep8_batch512_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/disagg-3p1d-dep8-b512-eplb0-mtp0-c512.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h200-fp8/8k1k/variants.yaml:override_disagg_3p1d_dep8_b512_eplb0_mtp0_c512" decode: num-worker: 1 tp: 8 @@ -1878,7 +1878,7 @@ dsr1-fp8-h100-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h100-fp8/8k1k/mtp/ctx1_gen3_tep16_batch1_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b1-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep16_b1_eplb0_mtp3" decode: num-worker: 3 tp: 16 @@ -1893,7 +1893,7 @@ dsr1-fp8-h100-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h100-fp8/8k1k/mtp/ctx1_gen3_tep16_batch2_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b2-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep16_b2_eplb0_mtp3" decode: num-worker: 3 tp: 16 @@ -1908,7 +1908,7 @@ dsr1-fp8-h100-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h100-fp8/8k1k/mtp/ctx1_gen3_tep16_batch8_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b8-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep16_b8_eplb0_mtp3" decode: num-worker: 3 tp: 16 @@ -1923,7 +1923,7 @@ dsr1-fp8-h100-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h100-fp8/8k1k/mtp/ctx1_gen1_dep16_batch4_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p1d-dep16-b4-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p1d_dep16_b4_eplb0_mtp3" decode: num-worker: 1 tp: 16 @@ -1940,7 +1940,7 @@ dsr1-fp8-h100-dynamo-trt: # dp-attn: true # additional-settings: # # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h100-fp8/8k1k/mtp/ctx1_gen2_tep16_batch32_eplb0_mtp3.yaml - # - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p2d-tep16-b32-eplb0-mtp3.yaml" + # - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p2d_tep16_b32_eplb0_mtp3" # decode: # num-worker: 2 # tp: 16 @@ -1955,7 +1955,7 @@ dsr1-fp8-h100-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h100-fp8/8k1k/mtp/ctx2_gen1_dep16_batch8_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-2p1d-dep16-b8-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_2p1d_dep16_b8_eplb0_mtp3" decode: num-worker: 1 tp: 16 @@ -1970,7 +1970,7 @@ dsr1-fp8-h100-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h100-fp8/8k1k/stp/ctx1_gen3_tep16_batch1_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b1-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep16_b1_eplb0_mtp0" decode: num-worker: 3 tp: 16 @@ -1984,7 +1984,7 @@ dsr1-fp8-h100-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h100-fp8/8k1k/stp/ctx1_gen3_tep16_batch2_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b2-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep16_b2_eplb0_mtp0" decode: num-worker: 3 tp: 16 @@ -1998,7 +1998,7 @@ dsr1-fp8-h100-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h100-fp8/8k1k/stp/ctx1_gen3_tep16_batch8_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p3d-tep16-b8-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep16_b8_eplb0_mtp0" decode: num-worker: 3 tp: 16 @@ -2012,7 +2012,7 @@ dsr1-fp8-h100-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h100-fp8/8k1k/stp/ctx1_gen2_tep16_batch64_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-1p2d-tep16-b64-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_1p2d_tep16_b64_eplb0_mtp0" decode: num-worker: 2 tp: 16 @@ -2026,7 +2026,7 @@ dsr1-fp8-h100-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/h100-fp8/8k1k/stp/ctx2_gen1_dep16_batch16_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/disagg-2p1d-dep16-b16-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/h100-fp8/8k1k/variants.yaml:override_disagg_2p1d_dep16_b16_eplb0_mtp0" decode: num-worker: 1 tp: 16 @@ -2057,7 +2057,7 @@ dsr1-fp8-h100-dynamo-sglang: # ep: 1 # dp-attn: false # additional-settings: - # - "CONFIG_FILE=recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-b128-c1x2x4x8x16x32x64x128-stp.yaml" + # - "CONFIG_FILE=recipes/dsr1/sglang/h100-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp16_d_tp16_b128_c1x2x4x8x16x32x64x128_stp" # decode: # num-worker: 1 # tp: 16 @@ -2071,7 +2071,7 @@ dsr1-fp8-h100-dynamo-sglang: # ep: 1 # dp-attn: false # additional-settings: - # - "CONFIG_FILE=recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-ep16-dp16-b64-c1x2x4x8x16x32x64-stp.yaml" + # - "CONFIG_FILE=recipes/dsr1/sglang/h100-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp16_d_tp16_ep16_dp16_b64_c1x2x4x8x16x32x64_stp" # decode: # num-worker: 1 # tp: 16 @@ -2086,7 +2086,7 @@ dsr1-fp8-h100-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-b128-c1x2x4x8x16x32x64x128-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/h100-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp16_d_tp16_b128_c1x2x4x8x16x32x64x128_mtp" decode: num-worker: 1 tp: 16 @@ -2101,7 +2101,7 @@ dsr1-fp8-h100-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/h100-fp8/8k1k/disagg-1p1d-p-tp16-d-tp16-ep16-dp16-b64-c1x2x4x8x16x32x64-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/h100-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp16_d_tp16_ep16_dp16_b64_c1x2x4x8x16x32x64_mtp" decode: num-worker: 1 tp: 16 @@ -2134,7 +2134,7 @@ dsr1-fp4-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp4/8k1k/mtp/ctx1_gen4_tep8_batch8_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b8-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b8_eplb0_mtp3" decode: num-worker: 4 tp: 8 @@ -2149,7 +2149,7 @@ dsr1-fp4-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp4/8k1k/mtp/ctx3_gen1_dep32_batch4_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-3p1d-dep32-b4-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_3p1d_dep32_b4_eplb0_mtp3" decode: num-worker: 1 tp: 32 @@ -2164,7 +2164,7 @@ dsr1-fp4-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp4/8k1k/mtp/ctx7_gen1_dep16_batch64_eplb256_mtp1.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-7p1d-dep16-b64-eplb256-mtp1.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_7p1d_dep16_b64_eplb256_mtp1" decode: num-worker: 1 tp: 16 @@ -2179,7 +2179,7 @@ dsr1-fp4-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp4/8k1k/mtp/ctx8_gen1_dep32_batch16_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-8p1d-dep32-b16-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_8p1d_dep32_b16_eplb0_mtp3" decode: num-worker: 1 tp: 32 @@ -2194,7 +2194,7 @@ dsr1-fp4-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp4/8k1k/mtp/ctx11_gen1_dep16_batch256_eplb256_mtp1.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-11p1d-dep16-b256-eplb256-mtp1.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_11p1d_dep16_b256_eplb256_mtp1" decode: num-worker: 1 tp: 16 @@ -2210,7 +2210,7 @@ dsr1-fp4-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp4/8k1k/stp/ctx1_gen4_tep8_batch16_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b16-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b16_eplb0_mtp0" decode: num-worker: 4 tp: 8 @@ -2224,7 +2224,7 @@ dsr1-fp4-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp4/8k1k/stp/ctx1_gen4_tep8_batch1_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp0" decode: num-worker: 4 tp: 8 @@ -2238,7 +2238,7 @@ dsr1-fp4-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp4/8k1k/stp/ctx2_gen1_dep32_batch8_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-2p1d-dep32-b8-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_2p1d_dep32_b8_eplb0_mtp0" decode: num-worker: 1 tp: 32 @@ -2252,7 +2252,7 @@ dsr1-fp4-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp4/8k1k/stp/ctx7_gen1_dep32_batch32_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-7p1d-dep32-b32-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_7p1d_dep32_b32_eplb0_mtp0" decode: num-worker: 1 tp: 32 @@ -2266,7 +2266,7 @@ dsr1-fp4-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp4/8k1k/stp/ctx8_gen1_dep16_batch128_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-8p1d-dep16-b128-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_8p1d_dep16_b128_eplb0_mtp0" decode: num-worker: 1 tp: 16 @@ -2280,7 +2280,7 @@ dsr1-fp4-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp4/8k1k/stp/ctx10_gen1_dep16_batch256_eplb256_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/disagg-10p1d-dep16-b256-eplb256-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp4/8k1k/variants.yaml:override_disagg_10p1d_dep16_b256_eplb256_mtp0" decode: num-worker: 1 tp: 16 @@ -2312,7 +2312,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/mtp/ctx3_gen1_dep8_batch64_eplb0_mtp3_666.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep8-b64-eplb0-mtp3-c666.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_3p1d_dep8_b64_eplb0_mtp3_c666" decode: num-worker: 1 tp: 8 @@ -2327,7 +2327,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/mtp/ctx5_gen1_dep16_batch32_eplb0_mtp3_666.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-5p1d-dep16-b32-eplb0-mtp3-c666.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_5p1d_dep16_b32_eplb0_mtp3_c666" decode: num-worker: 1 tp: 16 @@ -2342,7 +2342,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/mtp/ctx3_gen1_dep16_batch16_eplb0_mtp3_333.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep16-b16-eplb0-mtp3-c333.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_3p1d_dep16_b16_eplb0_mtp3_c333" decode: num-worker: 1 tp: 16 @@ -2357,7 +2357,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/mtp/ctx4_gen1_dep32_batch8_eplb0_mtp3_333.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-4p1d-dep32-b8-eplb0-mtp3-c333.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_4p1d_dep32_b8_eplb0_mtp3_c333" decode: num-worker: 1 tp: 32 @@ -2372,7 +2372,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/mtp/ctx2_gen1_dep32_batch2_eplb0_mtp3_90.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-2p1d-dep32-b2-eplb0-mtp3-c90.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_2p1d_dep32_b2_eplb0_mtp3_c90" decode: num-worker: 1 tp: 32 @@ -2387,7 +2387,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/mtp/ctx1_gen3_tep8_batch4_eplb0_mtp3_15.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b4-eplb0-mtp3-c15.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b4_eplb0_mtp3_c15" decode: num-worker: 3 tp: 8 @@ -2402,7 +2402,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/mtp/ctx1_gen3_tep8_batch2_eplb0_mtp3_6.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b2-eplb0-mtp3-c6.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b2_eplb0_mtp3_c6" decode: num-worker: 3 tp: 8 @@ -2417,7 +2417,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/stp/ctx5_gen1_dep16_batch64_eplb0_mtp0_1229.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-5p1d-dep16-b64-eplb0-mtp0-c1229.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_5p1d_dep16_b64_eplb0_mtp0_c1229" decode: num-worker: 1 tp: 16 @@ -2431,7 +2431,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/stp/ctx4_gen1_dep32_batch16_eplb0_mtp0_666.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-4p1d-dep32-b16-eplb0-mtp0-c666.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_4p1d_dep32_b16_eplb0_mtp0_c666" decode: num-worker: 1 tp: 32 @@ -2445,7 +2445,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/stp/ctx3_gen1_dep16_batch32_eplb0_mtp0_615.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-3p1d-dep16-b32-eplb0-mtp0-c615.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_3p1d_dep16_b32_eplb0_mtp0_c615" decode: num-worker: 1 tp: 16 @@ -2459,7 +2459,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/stp/ctx2_gen1_dep32_batch8_eplb0_mtp0_333.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-2p1d-dep32-b8-eplb0-mtp0-c333.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_2p1d_dep32_b8_eplb0_mtp0_c333" decode: num-worker: 1 tp: 32 @@ -2473,7 +2473,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/stp/ctx1_gen3_tep8_batch16_eplb0_mtp0_63.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0-c63.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b16_eplb0_mtp0_c63" decode: num-worker: 3 tp: 8 @@ -2487,7 +2487,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/stp/ctx1_gen3_tep8_batch4_eplb0_mtp0_18.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b4-eplb0-mtp0-c18.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b4_eplb0_mtp0_c18" decode: num-worker: 3 tp: 8 @@ -2501,7 +2501,7 @@ dsr1-fp8-gb200-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb200-fp8/8k1k/stp/ctx1_gen3_tep8_batch1_eplb0_mtp0_6.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/disagg-1p3d-tep8-b1-eplb0-mtp0-c6.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb200-fp8/8k1k/variants.yaml:override_disagg_1p3d_tep8_b1_eplb0_mtp0_c6" decode: num-worker: 3 tp: 8 @@ -2549,7 +2549,7 @@ dsr1-fp8-gb200-dynamo-sglang: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/gb200-fp8/8k1k/mid-curve.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/gb200-fp8/8k1k/disagg-5p1d-p-tp8-ep8-dp8-d-tp32-ep32-dp32-b8192-c512x1024x2048x6144-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_5p1d_p_tp8_ep8_dp8_d_tp32_ep32_dp32_b8192_c512x1024x2048x6144_stp" decode: num-worker: 1 tp: 32 @@ -2565,7 +2565,7 @@ dsr1-fp8-gb200-dynamo-sglang: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/gb200-fp8/8k1k/max_tpt.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/gb200-fp8/8k1k/disagg-6p1d-p-tp8-ep8-dp8-d-tp24-ep24-dp24-b8192-c2048x4096x6144-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_6p1d_p_tp8_ep8_dp8_d_tp24_ep24_dp24_b8192_c2048x4096x6144_stp" decode: num-worker: 1 tp: 24 @@ -2597,7 +2597,7 @@ dsr1-fp8-gb300-dynamo-sglang: dp-attn: false additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/gb300-fp8/8k1k/stp/low-latency.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c4x8-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_b128_c4x8_stp" decode: num-worker: 1 tp: 4 @@ -2613,7 +2613,7 @@ dsr1-fp8-gb300-dynamo-sglang: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/gb300-fp8/8k1k/stp/mid.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-5p1d-p-tp8-ep8-dp8-d-tp32-ep32-dp32-b45000-c128x256x512x1024-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_5p1d_p_tp8_ep8_dp8_d_tp32_ep32_dp32_b45000_c128x256x512x1024_stp" decode: num-worker: 1 tp: 32 @@ -2629,7 +2629,7 @@ dsr1-fp8-gb300-dynamo-sglang: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/gb300-fp8/8k1k/stp/max.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp8-ep8-dp8-d-tp24-ep24-dp24-b45000-c2048x4096-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_6p1d_p_tp8_ep8_dp8_d_tp24_ep24_dp24_b45000_c2048x4096_stp" decode: num-worker: 1 tp: 24 @@ -2661,7 +2661,7 @@ dsr1-fp4-gb200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-1p4d-p-tp4-d-tp4-c4x8-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/gb200-fp4/8k1k/variants.yaml:override_disagg_1p4d_p_tp4_d_tp4_c4x8_stp" decode: num-worker: 4 tp: 4 @@ -2677,7 +2677,7 @@ dsr1-fp4-gb200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-6p1d-p-tp4-d-tp48-ep48-dp48-b16384-c512x2048x4096-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/gb200-fp4/8k1k/variants.yaml:override_disagg_6p1d_p_tp4_d_tp48_ep48_dp48_b16384_c512x2048x4096_stp" decode: num-worker: 1 tp: 48 @@ -2693,7 +2693,7 @@ dsr1-fp4-gb200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/gb200-fp4/8k1k/disagg-10p1d-p-tp4-d-tp32-ep32-dp32-b16384-c2048-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/gb200-fp4/8k1k/variants.yaml:override_disagg_10p1d_p_tp4_d_tp32_ep32_dp32_b16384_c2048_stp" decode: num-worker: 1 tp: 32 @@ -2726,7 +2726,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/mtp/ctx1_gen3_tep8_batch8_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p3d-tep8-b8-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p3d_tep8_b8_eplb0_mtp3" decode: num-worker: 3 tp: 8 @@ -2741,7 +2741,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/mtp/ctx1_gen4_tep8_batch1_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp3" decode: num-worker: 4 tp: 8 @@ -2756,7 +2756,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/mtp/ctx1_gen4_tep8_batch4_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b4_eplb0_mtp3" decode: num-worker: 4 tp: 8 @@ -2771,7 +2771,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/mtp/ctx4_gen1_dep32_batch4_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-4p1d-dep32-b4-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_4p1d_dep32_b4_eplb0_mtp3" decode: num-worker: 1 tp: 32 @@ -2786,7 +2786,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/mtp/ctx8_gen1_dep32_batch8_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-8p1d-dep32-b8-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_8p1d_dep32_b8_eplb0_mtp3" decode: num-worker: 1 tp: 32 @@ -2801,7 +2801,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/mtp/ctx10_gen1_dep8_batch256_eplb0_mtp1.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp1.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_10p1d_dep8_b256_eplb0_mtp1" decode: num-worker: 1 tp: 8 @@ -2816,7 +2816,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/mtp/ctx10_gen1_dep16_batch32_eplb0_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep16-b32-eplb0-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_10p1d_dep16_b32_eplb0_mtp3" decode: num-worker: 1 tp: 16 @@ -2831,7 +2831,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/mtp/ctx13_gen1_dep16_batch64_eplb256_mtp3.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-13p1d-dep16-b64-eplb256-mtp3.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_13p1d_dep16_b64_eplb256_mtp3" decode: num-worker: 1 tp: 16 @@ -2846,7 +2846,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/stp/ctx1_gen3_tep8_batch16_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p3d-tep8-b16-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p3d_tep8_b16_eplb0_mtp0" decode: num-worker: 3 tp: 8 @@ -2860,7 +2860,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/stp/ctx1_gen4_tep8_batch1_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp0" decode: num-worker: 4 tp: 8 @@ -2874,7 +2874,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/stp/ctx1_gen4_tep8_batch2_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b2-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b2_eplb0_mtp0" decode: num-worker: 4 tp: 8 @@ -2888,7 +2888,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/stp/ctx1_gen5_tep4_batch4_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-1p5d-tep4-b4-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p5d_tep4_b4_eplb0_mtp0" decode: num-worker: 5 tp: 4 @@ -2902,7 +2902,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/stp/ctx7_gen1_dep32_batch16_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep32-b16-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_7p1d_dep32_b16_eplb0_mtp0" decode: num-worker: 1 tp: 32 @@ -2916,7 +2916,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/stp/ctx9_gen1_dep16_batch64_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-9p1d-dep16-b64-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_9p1d_dep16_b64_eplb0_mtp0" decode: num-worker: 1 tp: 16 @@ -2930,7 +2930,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/stp/ctx11_gen3_dep4_batch256_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-11p3d-dep4-b256-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_11p3d_dep4_b256_eplb0_mtp0" decode: num-worker: 3 tp: 4 @@ -2944,7 +2944,7 @@ dsr1-fp4-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp4/8k1k/stp/ctx14_gen1_dep16_batch128_eplb0_mtp0.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/disagg-14p1d-dep16-b128-eplb0-mtp0.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_14p1d_dep16_b128_eplb0_mtp0" decode: num-worker: 1 tp: 16 @@ -2975,7 +2975,7 @@ dsr1-fp4-gb300-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-1p4d-p-tp4-d-tp4-c4x8x32x64-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_p_tp4_d_tp4_c4x8x32x64_stp" decode: num-worker: 4 tp: 4 @@ -2991,7 +2991,7 @@ dsr1-fp4-gb300-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-6p1d-p-tp4-d-tp48-ep48-dp48-b16384-c512x2048x4096-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_6p1d_p_tp4_d_tp48_ep48_dp48_b16384_c512x2048x4096_stp" decode: num-worker: 1 tp: 48 @@ -3007,7 +3007,7 @@ dsr1-fp4-gb300-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/gb300-fp4/8k1k/disagg-10p1d-p-tp4-d-tp32-ep32-dp32-b16384-c2048-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_10p1d_p_tp4_d_tp32_ep32_dp32_b16384_c2048_stp" decode: num-worker: 1 tp: 32 @@ -3040,7 +3040,7 @@ dsr1-fp8-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp8/8k1k/mtp/ctx1_gen4_tep8_batch1_eplb0_mtp3_8.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3-c8.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp3_c8" decode: num-worker: 4 tp: 8 @@ -3055,7 +3055,7 @@ dsr1-fp8-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp8/8k1k/mtp/ctx1_gen4_tep8_batch4_eplb0_mtp3_24.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp3-c24.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tep8_b4_eplb0_mtp3_c24" decode: num-worker: 4 tp: 8 @@ -3070,7 +3070,7 @@ dsr1-fp8-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp8/8k1k/mtp/ctx6_gen1_dep32_batch8_eplb0_mtp3_333.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-6p1d-dep32-b8-eplb0-mtp3-c333.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_6p1d_dep32_b8_eplb0_mtp3_c333" decode: num-worker: 1 tp: 32 @@ -3085,7 +3085,7 @@ dsr1-fp8-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp8/8k1k/mtp/ctx8_gen1_dep16_batch32_eplb0_mtp3_666.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-8p1d-dep16-b32-eplb0-mtp3-c666.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_8p1d_dep16_b32_eplb0_mtp3_c666" decode: num-worker: 1 tp: 16 @@ -3100,7 +3100,7 @@ dsr1-fp8-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp8/8k1k/mtp/ctx10_gen1_dep16_batch64_eplb0_mtp1_1229.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-10p1d-dep16-b64-eplb0-mtp1-c1229.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_10p1d_dep16_b64_eplb0_mtp1_c1229" decode: num-worker: 1 tp: 16 @@ -3115,7 +3115,7 @@ dsr1-fp8-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp8/8k1k/mtp/ctx7_gen1_dep8_batch128_eplb0_mtp1_1229.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep8-b128-eplb0-mtp1-c1229.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_7p1d_dep8_b128_eplb0_mtp1_c1229" decode: num-worker: 1 tp: 8 @@ -3130,7 +3130,7 @@ dsr1-fp8-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp8/8k1k/stp/ctx1_gen4_tep8_batch1_eplb0_mtp0_4.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0-c4.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp0_c4" decode: num-worker: 4 tp: 8 @@ -3144,7 +3144,7 @@ dsr1-fp8-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp8/8k1k/stp/ctx1_gen4_tep8_batch4_eplb0_mtp0_24.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp0-c24.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tep8_b4_eplb0_mtp0_c24" decode: num-worker: 4 tp: 8 @@ -3158,7 +3158,7 @@ dsr1-fp8-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp8/8k1k/stp/ctx1_gen4_tep8_batch8_eplb0_mtp0_36.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-1p4d-tep8-b8-eplb0-mtp0-c36.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_1p4d_tep8_b8_eplb0_mtp0_c36" decode: num-worker: 4 tp: 8 @@ -3172,7 +3172,7 @@ dsr1-fp8-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp8/8k1k/stp/ctx6_gen1_dep32_batch16_eplb0_mtp0_512.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-6p1d-dep32-b16-eplb0-mtp0-c512.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_6p1d_dep32_b16_eplb0_mtp0_c512" decode: num-worker: 1 tp: 32 @@ -3186,7 +3186,7 @@ dsr1-fp8-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp8/8k1k/stp/ctx4_gen1_dep16_batch32_eplb0_mtp0_666.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-4p1d-dep16-b32-eplb0-mtp0-c666.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_4p1d_dep16_b32_eplb0_mtp0_c666" decode: num-worker: 1 tp: 16 @@ -3200,7 +3200,7 @@ dsr1-fp8-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp8/8k1k/stp/ctx7_gen1_dep16_batch64_eplb0_mtp0_1229.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep16-b64-eplb0-mtp0-c1229.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_7p1d_dep16_b64_eplb0_mtp0_c1229" decode: num-worker: 1 tp: 16 @@ -3214,7 +3214,7 @@ dsr1-fp8-gb300-dynamo-trt: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/trtllm/gb300-fp8/8k1k/stp/ctx7_gen1_dep8_batch256_eplb0_mtp0_2151.yaml - - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/disagg-7p1d-dep8-b256-eplb0-mtp0-c2151.yaml" + - "CONFIG_FILE=recipes/dsr1/trtllm/gb300-fp8/8k1k/variants.yaml:override_disagg_7p1d_dep8_b256_eplb0_mtp0_c2151" decode: num-worker: 1 tp: 8 @@ -3245,7 +3245,7 @@ dsr1-fp8-h200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs4-1p7d-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs4_1p7d_stp" decode: num-worker: 7 tp: 8 @@ -3260,7 +3260,7 @@ dsr1-fp8-h200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs8-1p6d-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs8_1p6d_stp" decode: num-worker: 6 tp: 8 @@ -3275,7 +3275,7 @@ dsr1-fp8-h200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs16-1p3d-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs16_1p3d_stp" decode: num-worker: 3 tp: 8 @@ -3290,7 +3290,7 @@ dsr1-fp8-h200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs64-2p3d-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs64_2p3d_stp" decode: num-worker: 3 tp: 8 @@ -3305,7 +3305,7 @@ dsr1-fp8-h200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs128-1p1d-dep-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs128_1p1d_dep_stp" decode: num-worker: 1 tp: 8 @@ -3320,7 +3320,7 @@ dsr1-fp8-h200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs4-1p7d-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs4_1p7d_mtp" decode: num-worker: 7 tp: 8 @@ -3335,7 +3335,7 @@ dsr1-fp8-h200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs8-1p6d-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs8_1p6d_mtp" decode: num-worker: 6 tp: 8 @@ -3350,7 +3350,7 @@ dsr1-fp8-h200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs16-1p3d-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs16_1p3d_mtp" decode: num-worker: 3 tp: 8 @@ -3365,7 +3365,7 @@ dsr1-fp8-h200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs64-2p3d-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs64_2p3d_mtp" decode: num-worker: 3 tp: 8 @@ -3380,7 +3380,7 @@ dsr1-fp8-h200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/disagg-bs128-1p1d-dep-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/h200-fp8/8k1k/variants.yaml:override_disagg_bs128_1p1d_dep_mtp" decode: num-worker: 1 tp: 8 @@ -3492,7 +3492,7 @@ dsr1-fp8-b200-dynamo-sglang: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_stp_lowlat_0.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p3d-p-tp8-dp8-d-tp8-b32-c128-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p3d_p_tp8_dp8_d_tp8_b32_c128_stp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 3 @@ -3507,7 +3507,7 @@ dsr1-fp8-b200-dynamo-sglang: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_stp_lowlat_1.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p4d-p-tp8-dp8-d-tp8-b32-c128-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p4d_p_tp8_dp8_d_tp8_b32_c128_stp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 4 @@ -3522,7 +3522,7 @@ dsr1-fp8-b200-dynamo-sglang: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_stp_lowlat_2.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p6d-p-tp8-dp8-d-tp8-b22-c8x16x32x64x128-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p6d_p_tp8_dp8_d_tp8_b22_c8x16x32x64x128_stp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 6 @@ -3538,7 +3538,7 @@ dsr1-fp8-b200-dynamo-sglang: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_stp_maxtpt_0.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p2d-p-tp8-dp8-d-tp8-ep8-dp8-b128-c288-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p2d_p_tp8_dp8_d_tp8_ep8_dp8_b128_c288_stp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 2 @@ -3553,7 +3553,7 @@ dsr1-fp8-b200-dynamo-sglang: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_stp_maxtpt_1.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p1d-p-tp8-dp8-d-tp8-ep8-dp8-b256-c160x288-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp8_dp8_d_tp8_ep8_dp8_b256_c160x288_stp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 1 @@ -3568,7 +3568,7 @@ dsr1-fp8-b200-dynamo-sglang: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_stp_maxtpt_2.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-2p1d-p-tp8-dp8-d-tp8-ep8-dp8-b512-c512-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_2p1d_p_tp8_dp8_d_tp8_ep8_dp8_b512_c512_stp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 1 @@ -3583,7 +3583,7 @@ dsr1-fp8-b200-dynamo-sglang: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_stp_maxtpt_3.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-3p1d-p-tp8-dp8-d-tp8-ep8-dp8-b1024-c1024-stp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_3p1d_p_tp8_dp8_d_tp8_ep8_dp8_b1024_c1024_stp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 1 @@ -3616,7 +3616,7 @@ dsr1-fp8-b200-dynamo-sglang-mtp: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_mtp_lowlat_0.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p3d-p-tp8-dp8-d-tp8-b32-c128-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p3d_p_tp8_dp8_d_tp8_b32_c128_mtp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 3 @@ -3632,7 +3632,7 @@ dsr1-fp8-b200-dynamo-sglang-mtp: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_mtp_lowlat_1.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p4d-p-tp8-dp8-d-tp8-b32-c128-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p4d_p_tp8_dp8_d_tp8_b32_c128_mtp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 4 @@ -3648,7 +3648,7 @@ dsr1-fp8-b200-dynamo-sglang-mtp: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_mtp_lowlat_2.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p6d-p-tp8-dp8-d-tp8-b22-c8x16x32x64x128-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p6d_p_tp8_dp8_d_tp8_b22_c8x16x32x64x128_mtp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 6 @@ -3665,7 +3665,7 @@ dsr1-fp8-b200-dynamo-sglang-mtp: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_mtp_maxtpt_0.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p2d-p-tp8-dp8-d-tp8-ep8-dp8-b128-c288-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p2d_p_tp8_dp8_d_tp8_ep8_dp8_b128_c288_mtp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 2 @@ -3681,7 +3681,7 @@ dsr1-fp8-b200-dynamo-sglang-mtp: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_mtp_maxtpt_1.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-1p1d-p-tp8-dp8-d-tp8-ep8-dp8-b256-c160x288-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp8_dp8_d_tp8_ep8_dp8_b256_c160x288_mtp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 1 @@ -3697,7 +3697,7 @@ dsr1-fp8-b200-dynamo-sglang-mtp: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_mtp_maxtpt_2.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-2p1d-p-tp8-dp8-d-tp8-ep8-dp8-b512-c512-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_2p1d_p_tp8_dp8_d_tp8_ep8_dp8_b512_c512_mtp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 1 @@ -3713,7 +3713,7 @@ dsr1-fp8-b200-dynamo-sglang-mtp: dp-attn: true additional-settings: # https://github.com/NVIDIA/srt-slurm/blob/sa-submission-q2-2026/recipes/b200-fp8/8k1k_mtp_maxtpt_3.yaml - - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/disagg-3p1d-p-tp8-dp8-d-tp8-ep8-dp8-b1024-c1024-mtp.yaml" + - "CONFIG_FILE=recipes/dsr1/sglang/b200-fp8/8k1k/variants.yaml:override_disagg_3p1d_p_tp8_dp8_d_tp8_ep8_dp8_b1024_c1024_mtp" router: { name: dynamo-router, version: "0.9.1" } decode: num-worker: 1 @@ -3899,7 +3899,7 @@ qwen3.5-fp8-gb200-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-tp4-tp4-stp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_1p1d_tp4_tp4_stp" decode: num-worker: 1 tp: 4 @@ -3914,7 +3914,7 @@ qwen3.5-fp8-gb200-dynamo-sglang: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-4p1d-dep4-dep16-stp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_4p1d_dep4_dep16_stp" decode: num-worker: 1 tp: 16 @@ -3929,7 +3929,7 @@ qwen3.5-fp8-gb200-dynamo-sglang: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-8p1d-dep4-dep16-stp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_8p1d_dep4_dep16_stp" decode: num-worker: 1 tp: 16 @@ -3996,7 +3996,7 @@ qwen3.5-fp4-gb300-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c1x4x8x16x32x64x256-stp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_b128_c1x4x8x16x32x64x256_stp" decode: num-worker: 1 tp: 4 @@ -4013,7 +4013,7 @@ qwen3.5-fp4-gb300-dynamo-sglang: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-5p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b4096-c2048-stp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_5p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b4096_c2048_stp" decode: num-worker: 1 tp: 16 @@ -4030,7 +4030,7 @@ qwen3.5-fp4-gb300-dynamo-sglang: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b5120-c5120-stp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_6p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b5120_c5120_stp" decode: num-worker: 1 tp: 16 @@ -4047,7 +4047,7 @@ qwen3.5-fp4-gb300-dynamo-sglang: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b5120-c5120-stp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/8k1k/variants.yaml:override_disagg_7p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b5120_c5120_stp" decode: num-worker: 1 tp: 16 @@ -4077,7 +4077,7 @@ qwen3.5-fp4-gb300-dynamo-trt: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p2d-tep8-b16-eplb0-mtp0-c42.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p2d_tep8_b16_eplb0_mtp0_c42" decode: num-worker: 2 tp: 8 @@ -4091,7 +4091,7 @@ qwen3.5-fp4-gb300-dynamo-trt: ep: 2 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp0-c8.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp0_c8" decode: num-worker: 4 tp: 8 @@ -4105,7 +4105,7 @@ qwen3.5-fp4-gb300-dynamo-trt: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b4-eplb0-mtp0-c24.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b4_eplb0_mtp0_c24" decode: num-worker: 4 tp: 8 @@ -4119,7 +4119,7 @@ qwen3.5-fp4-gb300-dynamo-trt: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p1d-dep16-b16-eplb0-mtp0-c282.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_4p1d_dep16_b16_eplb0_mtp0_c282" decode: num-worker: 1 tp: 16 @@ -4133,7 +4133,7 @@ qwen3.5-fp4-gb300-dynamo-trt: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p3d-tep8-b64-eplb0-mtp0-c210.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_4p3d_tep8_b64_eplb0_mtp0_c210" decode: num-worker: 3 tp: 8 @@ -4147,7 +4147,7 @@ qwen3.5-fp4-gb300-dynamo-trt: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep16-b32-eplb0-mtp0-c563.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_7p1d_dep16_b32_eplb0_mtp0_c563" decode: num-worker: 1 tp: 16 @@ -4161,7 +4161,7 @@ qwen3.5-fp4-gb300-dynamo-trt: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-10p1d-dep8-b256-eplb0-mtp0-c2150.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_10p1d_dep8_b256_eplb0_mtp0_c2150" decode: num-worker: 1 tp: 8 @@ -4175,7 +4175,7 @@ qwen3.5-fp4-gb300-dynamo-trt: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-11p1d-dep16-b64-eplb0-mtp0-c1076.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_11p1d_dep16_b64_eplb0_mtp0_c1076" decode: num-worker: 1 tp: 16 @@ -4189,7 +4189,7 @@ qwen3.5-fp4-gb300-dynamo-trt: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-16p1d-dep16-b128-eplb0-mtp0-c2253.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_16p1d_dep16_b128_eplb0_mtp0_c2253" decode: num-worker: 1 tp: 16 @@ -4203,7 +4203,7 @@ qwen3.5-fp4-gb300-dynamo-trt: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-24p1d-dep16-b512-eplb0-mtp0-c8192-stp-sweep.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_24p1d_dep16_b512_eplb0_mtp0_c8192_stp_sweep" decode: num-worker: 1 tp: 16 @@ -4238,7 +4238,7 @@ qwen3.5-fp4-gb300-dynamo-trt-mtp: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p2d-tep8-b8-eplb0-mtp3-c20.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p2d_tep8_b8_eplb0_mtp3_c20" decode: num-worker: 2 tp: 8 @@ -4253,7 +4253,7 @@ qwen3.5-fp4-gb300-dynamo-trt-mtp: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b1-eplb0-mtp3-c12.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b1_eplb0_mtp3_c12" decode: num-worker: 4 tp: 8 @@ -4268,7 +4268,7 @@ qwen3.5-fp4-gb300-dynamo-trt-mtp: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-1p4d-tep8-b2-eplb0-mtp3-c8.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_1p4d_tep8_b2_eplb0_mtp3_c8" decode: num-worker: 4 tp: 8 @@ -4283,7 +4283,7 @@ qwen3.5-fp4-gb300-dynamo-trt-mtp: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-4p3d-tep8-b32-eplb0-mtp3-c126.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_4p3d_tep8_b32_eplb0_mtp3_c126" decode: num-worker: 3 tp: 8 @@ -4298,7 +4298,7 @@ qwen3.5-fp4-gb300-dynamo-trt-mtp: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-5p1d-dep16-b8-eplb0-mtp3-c154.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_5p1d_dep16_b8_eplb0_mtp3_c154" decode: num-worker: 1 tp: 16 @@ -4313,7 +4313,7 @@ qwen3.5-fp4-gb300-dynamo-trt-mtp: ep: 2 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-7p1d-dep16-b32-eplb0-mtp3-c666.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_7p1d_dep16_b32_eplb0_mtp3_c666" decode: num-worker: 1 tp: 16 @@ -4328,7 +4328,7 @@ qwen3.5-fp4-gb300-dynamo-trt-mtp: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-11p1d-dep8-b128-eplb0-mtp3-c1229.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_11p1d_dep8_b128_eplb0_mtp3_c1229" decode: num-worker: 1 tp: 8 @@ -4343,7 +4343,7 @@ qwen3.5-fp4-gb300-dynamo-trt-mtp: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-17p2d-dep8-b64-eplb0-mtp3-c1126.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_17p2d_dep8_b64_eplb0_mtp3_c1126" decode: num-worker: 2 tp: 8 @@ -4358,7 +4358,7 @@ qwen3.5-fp4-gb300-dynamo-trt-mtp: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-26p1d-dep16-b256-eplb0-mtp2-c4301.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_26p1d_dep16_b256_eplb0_mtp2_c4301" decode: num-worker: 1 tp: 16 @@ -4373,7 +4373,7 @@ qwen3.5-fp4-gb300-dynamo-trt-mtp: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/disagg-24p1d-dep16-b512-eplb0-mtp0-c8192-mtp-sweep.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/8k1k/variants.yaml:override_disagg_24p1d_dep16_b512_eplb0_mtp0_c8192_mtp_sweep" decode: num-worker: 1 tp: 16 @@ -4407,7 +4407,7 @@ qwen3.5-fp4-gb300-dynamo-trt-agentic-disagg: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p1d-dep1-tep2-c44-b8-mtp-kvoffload.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep1_tep2_c44_b8_mtp_kvoffload" decode: num-worker: 1 tp: 2 @@ -4424,7 +4424,7 @@ qwen3.5-fp4-gb300-dynamo-trt-agentic-disagg: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p7d-dep4-tep8-c7-b1-mtp-kvoffload.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p7d_dep4_tep8_c7_b1_mtp_kvoffload" decode: num-worker: 7 tp: 8 @@ -4441,7 +4441,7 @@ qwen3.5-fp4-gb300-dynamo-trt-agentic-disagg: ep: 1 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-2p2d-dep1-tep2-c52-b4-mtp-kvoffload.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p2d_dep1_tep2_c52_b4_mtp_kvoffload" decode: num-worker: 2 tp: 2 @@ -4458,7 +4458,7 @@ qwen3.5-fp4-gb300-dynamo-trt-agentic-disagg: ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-2p3d-tep2-tep8-c96-b128-mtp-kvoffload.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p3d_tep2_tep8_c96_b128_mtp_kvoffload" decode: num-worker: 3 tp: 8 @@ -4475,7 +4475,7 @@ qwen3.5-fp4-gb300-dynamo-trt-agentic-disagg: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-3p1d-dep4-dep16-c565-b8-mtp-kvoffload.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_3p1d_dep4_dep16_c565_b8_mtp_kvoffload" decode: num-worker: 1 tp: 16 @@ -4492,7 +4492,7 @@ qwen3.5-fp4-gb300-dynamo-trt-agentic-disagg: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/agentx/disagg-3p2d-dep4-dep4-c704-b32-mtp-kvoffload.yaml" + - "CONFIG_FILE=recipes/qwen3.5/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_3p2d_dep4_dep4_c704_b32_mtp_kvoffload" decode: num-worker: 2 tp: 4 @@ -4523,7 +4523,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c1-mtp-hicache-jid2530006.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c1_mtp_hicache_jid2530006" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: hicache } @@ -4535,7 +4535,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c24-mtp-hicache-jid2530012.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c24_mtp_hicache_jid2530012" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: hicache } @@ -4547,7 +4547,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c32-mtp-hicache-jid2530013.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c32_mtp_hicache_jid2530013" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: hicache } @@ -4559,7 +4559,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c40-mtp-hicache-jid2530015.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c40_mtp_hicache_jid2530015" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: hicache } @@ -4571,7 +4571,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c48-mtp-hicache-jid2530017.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c48_mtp_hicache_jid2530017" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: hicache } @@ -4583,7 +4583,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c52-mtp-hicache-jid2527406.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c52_mtp_hicache_jid2527406" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: hicache } @@ -4595,7 +4595,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c64-mtp-hicache-jid2527410.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c64_mtp_hicache_jid2527410" # The seven disaggregated frontier points use six TP4 shapes plus one TP2 # shape. Stable X-Dynamo-Session-ID affinity replaces the removed conv-aware # routing message path. @@ -4624,7 +4624,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c8-mtp-hicache-session-jid2530030.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c8_mtp_hicache_session_jid2530030" decode: num-worker: 1 tp: 4 @@ -4640,7 +4640,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c16-mtp-hicache-session-jid2530027.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c16_mtp_hicache_session_jid2530027" decode: num-worker: 1 tp: 4 @@ -4656,7 +4656,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c32-mtp-hicache-session-jid2530028.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c32_mtp_hicache_session_jid2530028" decode: num-worker: 1 tp: 4 @@ -4672,7 +4672,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c64-mtp-hicache-session-jid2530029.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c64_mtp_hicache_session_jid2530029" decode: num-worker: 1 tp: 4 @@ -4688,7 +4688,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c96-mtp-hicache-session-jid2527409.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c96_mtp_hicache_session_jid2527409" decode: num-worker: 1 tp: 4 @@ -4704,7 +4704,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp4-tp4-c128-mtp-hicache-session-jid2527417.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c128_mtp_hicache_session_jid2527417" decode: num-worker: 1 tp: 4 @@ -4720,7 +4720,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-1p1d-tp2-tp2-c72-mtp-hicache-session-jid2527415.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp2_tp2_c72_mtp_hicache_session_jid2527415" decode: num-worker: 1 tp: 2 @@ -4757,7 +4757,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-pp-pareto: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-3p2d-pp4-dep4-c704-mtp-hicache-nightly-c20260831.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_3p2d_pp4_dep4_c704_mtp_hicache_nightly_c20260831" decode: num-worker: 2 tp: 4 @@ -4775,7 +4775,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-pp-pareto: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/disagg-3p4d-pp4-dep4-c565-mtp-hicache-nightly-c20260831.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_3p4d_pp4_dep4_c565_mtp_hicache_nightly_c20260831" decode: num-worker: 4 tp: 4 @@ -4808,7 +4808,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg-pareto: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c44-b2-mtp-hicache-nightly-c20260831.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c44_b2_mtp_hicache_nightly_c20260831" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: hicache } @@ -4821,7 +4821,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg-pareto: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp2-c44-b1-mtp-hicache-nightly-c20260831.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp2_c44_b1_mtp_hicache_nightly_c20260831" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: hicache } @@ -4834,7 +4834,7 @@ qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg-pareto: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/agg-tp8-c7-b1-mtp-hicache-nightly-c20260831.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp8_c7_b1_mtp_hicache_nightly_c20260831" qwen3.5-fp8-b300-sglang-agentic-mtp: image: lmsysorg/sglang:v0.5.16-cu130 @@ -4947,7 +4947,7 @@ dsv4-fp4-gb300-dynamo-vllm-agentic: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p6d-dep4-tp4.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p6d_dep4_tp4" decode: num-worker: 6 tp: 4 @@ -4963,7 +4963,7 @@ dsv4-fp4-gb300-dynamo-vllm-agentic: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p6d-dep4-tp4.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p6d_dep4_tp4" decode: num-worker: 6 tp: 4 @@ -4980,7 +4980,7 @@ dsv4-fp4-gb300-dynamo-vllm-agentic: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/disagg-4p1d-dep4-dep8-24-c4096.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_4p1d_dep4_dep8_24_c4096" decode: num-worker: 1 tp: 8 @@ -5233,7 +5233,7 @@ qwen3.5-fp4-gb200-dynamo-sglang-agentic-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp4_mtp" - spec-decoding: mtp kv-offloading: none conc-list: [2] @@ -5244,7 +5244,7 @@ qwen3.5-fp4-gb200-dynamo-sglang-agentic-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-no-symm.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp4_mtp_no_symm" - spec-decoding: mtp kv-offloading: none conc-list: [24] @@ -5255,7 +5255,7 @@ qwen3.5-fp4-gb200-dynamo-sglang-agentic-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp4-mtp-parity.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp4_mtp_parity" # Keep only the reproducible TP2/EP2 HiCache K5 transition point; C28 # was dominated by the published C28 result in the official PR sweep. - spec-decoding: mtp @@ -5269,7 +5269,7 @@ qwen3.5-fp4-gb200-dynamo-sglang-agentic-mtp: ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp2ep2_mtp_hicache" # Keep the published C32 interactivity anchor. The tighter capacity # layout becomes Pareto-relevant at C40 and remains deployable at C48. - spec-decoding: mtp @@ -5283,7 +5283,7 @@ qwen3.5-fp4-gb200-dynamo-sglang-agentic-mtp: ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache-k3-baseline.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp2ep2_mtp_hicache_k3_baseline" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: hicache } @@ -5295,7 +5295,7 @@ qwen3.5-fp4-gb200-dynamo-sglang-agentic-mtp: ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-tp2ep2-mtp-hicache-cap48.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp2ep2_mtp_hicache_cap48" minimaxm3-fp4-b300-vllm-agentic-mtp: image: vllm/vllm-openai:nightly-1dc464d42681d22f38caf1fdc1eb632dc4421c45 model: nvidia/MiniMax-M3-NVFP4 @@ -5340,8 +5340,8 @@ minimaxm3-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: ep: 4 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tep4-tp4-c1.yaml" - - "EVAL_CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tep4-tp4-c1-eval.yaml" + - "CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tep4_tp4_c1" + - "EVAL_CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tep4_tp4_c1_eval" decode: { num-worker: 1, tp: 4, ep: 1, dp-attn: false } - spec-decoding: mtp kv-offloading: dram @@ -5353,8 +5353,8 @@ minimaxm3-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-dep4-tp4-c24.yaml" - - "EVAL_CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-dep4-tp4-c24-eval.yaml" + - "CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p3d_dep4_tp4_c24" + - "EVAL_CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p3d_dep4_tp4_c24_eval" decode: { num-worker: 3, tp: 4, ep: 1, dp-attn: false } - spec-decoding: mtp kv-offloading: dram @@ -5366,8 +5366,8 @@ minimaxm3-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tp2-tp4-c20-c24.yaml" - - "EVAL_CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p1d-tp2-tp4-c20-c24-eval.yaml" + - "CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp2_tp4_c20_c24" + - "EVAL_CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp2_tp4_c20_c24_eval" decode: { num-worker: 1, tp: 4, ep: 1, dp-attn: false } - spec-decoding: mtp kv-offloading: dram @@ -5379,8 +5379,8 @@ minimaxm3-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-tp2-tp2-c48.yaml" - - "EVAL_CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-1p3d-tp2-tp2-c48-eval.yaml" + - "CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p3d_tp2_tp2_c48" + - "EVAL_CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p3d_tp2_tp2_c48_eval" decode: { num-worker: 3, tp: 2, ep: 1, dp-attn: false } - spec-decoding: mtp kv-offloading: dram @@ -5392,8 +5392,8 @@ minimaxm3-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-2p5d-tp2-tp2-c120.yaml" - - "EVAL_CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/disagg-2p5d-tp2-tp2-c120-eval.yaml" + - "CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p5d_tp2_tp2_c120" + - "EVAL_CONFIG_FILE=recipes/minimaxm3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p5d_tp2_tp2_c120_eval" decode: { num-worker: 5, tp: 2, ep: 1, dp-attn: false } minimaxm3-fp4-b300-trtllm-agentic-mtp: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc23.post1 @@ -5512,7 +5512,7 @@ minimaxm3-fp4-gb200-dynamo-vllm-agentic-agg-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-nightly-native.yaml" + - "CONFIG_FILE=recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_nightly_native" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: vllm-simple } @@ -5524,7 +5524,7 @@ minimaxm3-fp4-gb200-dynamo-vllm-agentic-agg-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp4-vllm-simple-nightly-native.yaml" + - "CONFIG_FILE=recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_vllm_simple_nightly_native" - spec-decoding: mtp kv-offloading: none conc-list: [1] @@ -5535,7 +5535,7 @@ minimaxm3-fp4-gb200-dynamo-vllm-agentic-agg-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/vllm/gb200-fp4/agentx/agg-tp8-nightly-native.yaml" + - "CONFIG_FILE=recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8_nightly_native" minimaxm3-fp4-gb200-dynamo-vllm-agentic-disagg-mtp: image: vllm/vllm-openai@sha256:b9104b7ef3048e42f79fba9ab5da06e5aff8164aca9968692ec2f569aaaf34c6 @@ -5563,7 +5563,7 @@ minimaxm3-fp4-gb200-dynamo-vllm-agentic-disagg-mtp: ep: 4 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-tp4-tp8-c1.yaml" + - "CONFIG_FILE=recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp8_c1" decode: { num-worker: 1, tp: 8, ep: 8, dp-attn: false } - spec-decoding: mtp kv-offloading: dram @@ -5575,7 +5575,7 @@ minimaxm3-fp4-gb200-dynamo-vllm-agentic-disagg-mtp: ep: 4 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p2d-tp4-tp4-c8-c16.yaml" + - "CONFIG_FILE=recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_1p2d_tp4_tp4_c8_c16" decode: { num-worker: 2, tp: 4, ep: 4, dp-attn: false } - spec-decoding: mtp kv-offloading: dram @@ -5587,7 +5587,7 @@ minimaxm3-fp4-gb200-dynamo-vllm-agentic-disagg-mtp: ep: 4 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/vllm/gb200-fp4/agentx/disagg-1p1d-tp4-tp4-c24.yaml" + - "CONFIG_FILE=recipes/minimaxm3/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_c24" decode: { num-worker: 1, tp: 4, ep: 4, dp-attn: false } minimaxm3-fp4-gb200-dynamo-trt-agentic-agg-mtp: @@ -5615,7 +5615,7 @@ minimaxm3-fp4-gb200-dynamo-trt-agentic-agg-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c5-b5-eagle3.yaml" + - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c5_b5_eagle3" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: native } @@ -5627,7 +5627,7 @@ minimaxm3-fp4-gb200-dynamo-trt-agentic-agg-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c10-b10-eagle3.yaml" + - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c10_b10_eagle3" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: native } @@ -5639,7 +5639,7 @@ minimaxm3-fp4-gb200-dynamo-trt-agentic-agg-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c15-b15-eagle3.yaml" + - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c15_b15_eagle3" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: native } @@ -5651,7 +5651,7 @@ minimaxm3-fp4-gb200-dynamo-trt-agentic-agg-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c20-b20-eagle3.yaml" + - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c20_b20_eagle3" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: native } @@ -5663,7 +5663,7 @@ minimaxm3-fp4-gb200-dynamo-trt-agentic-agg-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c25-b25-eagle3.yaml" + - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c25_b25_eagle3" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: native } @@ -5675,7 +5675,7 @@ minimaxm3-fp4-gb200-dynamo-trt-agentic-agg-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c30-b30-eagle3.yaml" + - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c30_b30_eagle3" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: native } @@ -5687,7 +5687,7 @@ minimaxm3-fp4-gb200-dynamo-trt-agentic-agg-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/agg-tp4-c40-b40-eagle3.yaml" + - "CONFIG_FILE=recipes/minimaxm3/trtllm/gb200-fp4/agentx/variants.yaml:override_agg_tp4_c40_b40_eagle3" dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-agg: image: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 @@ -5713,7 +5713,7 @@ dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8_mtp" - spec-decoding: mtp conc-list: [8, 12, 16, 20, 24, 28, 32, 36, 40, 44, 48, 52, 56, 60, 64, 68, 72, 76, 80] num-nodes: 2 @@ -5723,7 +5723,7 @@ dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-agg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/agg-dep8-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_agg_dep8_mtp" dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-disagg: image: vllm/vllm-openai:nightly-3ee2df30337a301164c46ae444b76ee67e71c106 model: deepseek-ai/DeepSeek-V4-Pro @@ -5747,7 +5747,7 @@ dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep8_mtp" decode: num-worker: 1 tp: 8 @@ -5762,7 +5762,7 @@ dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep8-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep8_mtp" decode: num-worker: 1 tp: 8 @@ -5793,7 +5793,7 @@ dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/agg-tp8-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_agg_tp8_mtp" - spec-decoding: mtp conc-list: [4] num-nodes: 2 @@ -5803,7 +5803,7 @@ dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/agg-tp8-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_agg_tp8_mtp" - spec-decoding: mtp conc-list: [1, 2, 4, 6, 8] num-nodes: 1 @@ -5813,7 +5813,7 @@ dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/agg-tp4-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_agg_tp4_mtp" dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f model: deepseek-ai/DeepSeek-V4-Pro @@ -5842,7 +5842,7 @@ dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep12-c1152-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep12_c1152_mtp" decode: num-worker: 1 tp: 12 @@ -5859,7 +5859,7 @@ dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep16-c1024-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep16_c1024_mtp" decode: num-worker: 1 tp: 16 @@ -5876,7 +5876,7 @@ dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep8-c256-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep4_dep8_c256_mtp" decode: num-worker: 1 tp: 8 @@ -5893,7 +5893,7 @@ dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep8-dep16-c512-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep16_c512_mtp" decode: num-worker: 1 tp: 16 @@ -5908,7 +5908,7 @@ dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p4d-dep4-tp8-c4-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p4d_dep4_tp8_c4_mtp" decode: num-worker: 4 tp: 8 @@ -5925,7 +5925,7 @@ dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep16-c128-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep4_dep16_c128_mtp" decode: num-worker: 1 tp: 16 @@ -5942,7 +5942,7 @@ dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/disagg-1p1d-dep4-dep16-c256-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep4_dep16_c256_mtp" decode: num-worker: 1 tp: 16 @@ -5978,7 +5978,7 @@ dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-c4-mtp3.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8_c4_mtp3" - spec-decoding: mtp conc-list: [4] num-nodes: 2 @@ -5988,7 +5988,7 @@ dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-c4-mtp3.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8_c4_mtp3" - spec-decoding: mtp conc-list: [8] num-nodes: 2 @@ -5998,7 +5998,7 @@ dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-c8-mtp3.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8_c8_mtp3" dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-disagg: image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f model: deepseek-ai/DeepSeek-V4-Pro @@ -6027,7 +6027,7 @@ dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep12-c576-mtp3.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep12_c576_mtp3" decode: num-worker: 1 tp: 12 @@ -6044,7 +6044,7 @@ dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/disagg-2p1d-dep8-dep16-c512-mtp3.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep16_c512_mtp3" decode: num-worker: 1 tp: 16 @@ -6061,7 +6061,7 @@ dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-c256-mtp3.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep8_c256_mtp3" decode: num-worker: 1 tp: 8 @@ -6078,7 +6078,7 @@ dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/disagg-1p1d-dep8-dep8-c128-mtp3.yaml" + - "CONFIG_FILE=recipes/dsv4/vllm/gb200-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep8_c128_mtp3" decode: num-worker: 1 tp: 8 @@ -6120,7 +6120,7 @@ kimik3-fp4-gb200-dynamo-vllm-agentic: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp16-latency.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp16_latency" # Balanced: multi_node_tep strategy, TEP16 across four GB200 nodes. # https://recipes.vllm.ai/moonshotai/Kimi-K3?hardware=gb200&nodes=4&strategy=multi_node_tep - spec-decoding: mtp @@ -6132,7 +6132,7 @@ kimik3-fp4-gb200-dynamo-vllm-agentic: ep: 16 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/agg-tep16-balanced.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tep16_balanced" # Throughput oriented: official multi_node_dep strategy, DEP16 across # four GB200 nodes (TP4 x DP4 = EP16, one local DP rank per node). # The recipe allows up to 3600s for full-context saturation warmup to drain. @@ -6146,7 +6146,7 @@ kimik3-fp4-gb200-dynamo-vllm-agentic: ep: 16 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/agg-dep16.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_dep16" # High-concurrency DEP16 with vLLM Simple CPU KV offloading. c384 # exercises 384 of the 393 AgentX trajectories and remains below the # aggregate max-num-seqs capacity of 512 (128 per DP rank). @@ -6161,7 +6161,7 @@ kimik3-fp4-gb200-dynamo-vllm-agentic: ep: 16 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/agg-dep16-vllm-simple-offload.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_dep16_vllm_simple_offload" # Kimi-K3 GB200 TP16/DCP16 profiles using Mooncake DRAM offload. kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-dcp16-agg: image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-75c2eef @@ -6188,7 +6188,7 @@ kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-dcp16-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/agg-dcp16-dspark4-maxseq2-mooncake.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_dcp16_dspark4_maxseq2_mooncake" kimik3-fp4-gb200-dynamo-vllm-agentic-mooncake-dcp16-agg: image: vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-75c2eef @@ -6214,7 +6214,7 @@ kimik3-fp4-gb200-dynamo-vllm-agentic-mooncake-dcp16-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/agg-dcp16-nospec-mooncake.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_dcp16_nospec_mooncake" dsv4-fp4-gb300-dynamo-trt-agentx: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc24 @@ -6241,7 +6241,7 @@ dsv4-fp4-gb300-dynamo-trt-agentx: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p4d-dep4-tep8-c4-b1-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p4d_dep4_tep8_c4_b1_mtp" - "SLURM_PARTITION=batch_1" decode: num-worker: 4 @@ -6258,7 +6258,7 @@ dsv4-fp4-gb300-dynamo-trt-agentx: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p6d-dep4-tep4-c24-b4-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p6d_dep4_tep4_c24_b4_mtp" - "SLURM_PARTITION=batch_1" decode: num-worker: 6 @@ -6275,7 +6275,7 @@ dsv4-fp4-gb300-dynamo-trt-agentx: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-1p1d-dep8-dep32-c388-b4-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep32_c388_b4_mtp" - "SLURM_PARTITION=batch_1" decode: num-worker: 1 @@ -6292,7 +6292,7 @@ dsv4-fp4-gb300-dynamo-trt-agentx: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-2p1d-dep8-dep32-c736-b8-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep32_c736_b8_mtp" - "SLURM_PARTITION=batch_1" decode: num-worker: 1 @@ -6309,7 +6309,7 @@ dsv4-fp4-gb300-dynamo-trt-agentx: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-3p1d-dep8-dep16-c1152-b32-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_3p1d_dep8_dep16_c1152_b32_mtp" - "SLURM_PARTITION=batch_1" decode: num-worker: 1 @@ -6326,7 +6326,7 @@ dsv4-fp4-gb300-dynamo-trt-agentx: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/trtllm/gb300-fp4/agentx/disagg-5p1d-dep8-dep16-c2626-b96-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_5p1d_dep8_dep16_c2626_b96_mtp" - "SLURM_PARTITION=batch_1" decode: num-worker: 1 @@ -6361,7 +6361,7 @@ kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-tp8pp2: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c16.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c16" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: mooncake, version: "0.3.11.post1" } @@ -6375,7 +6375,7 @@ kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-tp8pp2: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c32.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c32" - kv-offloading: dram kv-offload-backend: { name: mooncake, version: "0.3.11.post1" } conc-list: [48] @@ -6388,7 +6388,7 @@ kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-tp8pp2: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c48.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c48" - kv-offloading: dram kv-offload-backend: { name: mooncake, version: "0.3.11.post1" } conc-list: [72] @@ -6401,7 +6401,7 @@ kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-tp8pp2: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c72.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c72" - kv-offloading: dram kv-offload-backend: { name: mooncake, version: "0.3.11.post1" } conc-list: [96] @@ -6414,7 +6414,7 @@ kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-tp8pp2: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c96.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c96" dsv4-fp4-gb300-dynamo-sglang-agentic-agg: image: lmsysorg/sglang:v0.5.19-cu130@sha256:d6e7288627be8b02be88e4bba38e73f6d50e2826869f753c13a4c4385ab3eda9 model: deepseek-ai/DeepSeek-V4-Pro-0813 @@ -6437,7 +6437,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-tp8-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp8_mtp" - search-space: - spec-decoding: draft_model conc-list: [8] @@ -6448,7 +6448,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-tp4-mtp.yaml" + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml:override_agg_tp4_mtp" dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: image: lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21 model: deepseek-ai/DeepSeek-V4-Pro-0813 @@ -6474,7 +6474,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-1p1d-dep8-dep16-c480-mtp-kvoffload.yaml" + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep16_c480_mtp_kvoffload" decode: num-worker: 1 tp: 16 @@ -6490,7 +6490,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-2p1d-dep8-dep16-c960-mtp-kvoffload.yaml" + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep16_c960_mtp_kvoffload" decode: num-worker: 1 tp: 16 @@ -6506,7 +6506,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-3p1d-dep8-dep16-c1440-mtp-kvoffload.yaml" + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_3p1d_dep8_dep16_c1440_mtp_kvoffload" decode: num-worker: 1 tp: 16 @@ -6522,7 +6522,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-4p1d-dep8-dep16-c1920-mtp-kvoffload.yaml" + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/variants.yaml:override_disagg_4p1d_dep8_dep16_c1920_mtp_kvoffload" decode: num-worker: 1 tp: 16 @@ -6553,7 +6553,7 @@ qwen3.5-fp8-gb300-dynamo-sglang: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-tp4-tp4-stp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_1p1d_tp4_tp4_stp" decode: num-worker: 1 tp: 4 @@ -6568,7 +6568,7 @@ qwen3.5-fp8-gb300-dynamo-sglang: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-dep4-dep16-stp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_4p1d_dep4_dep16_stp" decode: num-worker: 1 tp: 16 @@ -6583,7 +6583,7 @@ qwen3.5-fp8-gb300-dynamo-sglang: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-dep4-dep16-stp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_8p1d_dep4_dep16_stp" decode: num-worker: 1 tp: 16 @@ -6951,7 +6951,7 @@ glm5.2-fp4-b200-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c1-mtp.yaml" + - "CONFIG_FILE=recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml:override_agg_tp8_c1_mtp" - spec-decoding: mtp conc-list: [4] kv-offloading: dram @@ -6963,7 +6963,7 @@ glm5.2-fp4-b200-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c4-mtp.yaml" + - "CONFIG_FILE=recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml:override_agg_tp8_c4_mtp" - spec-decoding: mtp conc-list: [8] kv-offloading: dram @@ -6975,7 +6975,7 @@ glm5.2-fp4-b200-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/glm5.2/sglang/b200-fp4/agentx/agg-tp8-c8-mtp.yaml" + - "CONFIG_FILE=recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml:override_agg_tp8_c8_mtp" glm5.2-fp4-b200-dynamo-sglang-agentic-disagg: image: lmsysorg/sglang:nightly-dev-20260910-00840301 @@ -7003,7 +7003,7 @@ glm5.2-fp4-b200-dynamo-sglang-agentic-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/glm5.2/sglang/b200-fp4/agentx/disagg-1p4d-dep8-tp4-c48-mtp.yaml" + - "CONFIG_FILE=recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml:override_disagg_1p4d_dep8_tp4_c48_mtp" decode: num-worker: 4 tp: 4 @@ -7020,7 +7020,7 @@ glm5.2-fp4-b200-dynamo-sglang-agentic-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/glm5.2/sglang/b200-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp.yaml" + - "CONFIG_FILE=recipes/glm5.2/sglang/b200-fp4/agentx/variants.yaml:override_disagg_1p1d_dep8_dep8_c64_mtp" decode: num-worker: 1 tp: 8 @@ -7117,7 +7117,7 @@ glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp: ep: 8 dp-attn: true additional-settings: - - CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-1p6d-dep8-tp4-c45-mtp.yaml + - CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml:override_disagg_1p6d_dep8_tp4_c45_mtp decode: num-worker: 6 tp: 4 @@ -7135,7 +7135,7 @@ glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp: ep: 8 dp-attn: true additional-settings: - - CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-1p4d-dep8-tp4-c48-mtp.yaml + - CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml:override_disagg_1p4d_dep8_tp4_c48_mtp decode: num-worker: 4 tp: 4 @@ -7153,7 +7153,7 @@ glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp: ep: 8 dp-attn: true additional-settings: - - CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-2p1d-dep8-dep16-c128-mtp.yaml + - CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml:override_disagg_2p1d_dep8_dep16_c128_mtp decode: num-worker: 1 tp: 16 @@ -7191,7 +7191,7 @@ glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp-agg: ep: 1 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c2-mtp.yaml + - CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp8_c2_mtp - spec-decoding: mtp kv-offloading: dram kv-offload-backend: @@ -7205,7 +7205,7 @@ glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp-agg: ep: 1 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c4-mtp.yaml + - CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp8_c4_mtp - spec-decoding: mtp kv-offloading: dram kv-offload-backend: @@ -7219,7 +7219,7 @@ glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp-agg: ep: 1 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/agg-tp8-c8-mtp.yaml + - CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/variants.yaml:override_agg_tp8_c8_mtp glm5.2-fp4-gb300-dynamo-sglang-agentic-agg: image: lmsysorg/sglang:nightly-dev-cu13-20260805-211ee642 model: nvidia/GLM-5.2-NVFP4 @@ -7374,7 +7374,7 @@ glm5.2-fp4-gb300-dynamo-trt-agentic-disagg-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p1d-tp8-c1-b1-mtp5.yaml" + - "CONFIG_FILE=recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tp8_c1_b1_mtp5" - "SLURM_PARTITION=batch_1" decode: num-worker: 1 @@ -7391,7 +7391,7 @@ glm5.2-fp4-gb300-dynamo-trt-agentic-disagg-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p4d-tep4-c30-b2-mtp5.yaml" + - "CONFIG_FILE=recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p4d_tep4_c30_b2_mtp5" - "SLURM_PARTITION=batch_1" decode: num-worker: 4 @@ -7408,7 +7408,7 @@ glm5.2-fp4-gb300-dynamo-trt-agentic-disagg-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-1p1d-tep8-c20-b5-mtp5.yaml" + - "CONFIG_FILE=recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_tep8_c20_b5_mtp5" - "SLURM_PARTITION=batch_1" decode: num-worker: 1 @@ -7425,7 +7425,7 @@ glm5.2-fp4-gb300-dynamo-trt-agentic-disagg-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-3p4d-tep4-c60-b5-mtp5.yaml" + - "CONFIG_FILE=recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_3p4d_tep4_c60_b5_mtp5" - "SLURM_PARTITION=batch_1" decode: num-worker: 4 @@ -7442,7 +7442,7 @@ glm5.2-fp4-gb300-dynamo-trt-agentic-disagg-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-5p1d-dep16-c260-b16-mtp3.yaml" + - "CONFIG_FILE=recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_5p1d_dep16_c260_b16_mtp3" - "SLURM_PARTITION=batch_1" decode: num-worker: 1 @@ -7459,7 +7459,7 @@ glm5.2-fp4-gb300-dynamo-trt-agentic-disagg-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/glm5.2/trtllm/gb300-fp4/agentx/disagg-4p1d-dep8-c227-b16-mtp3.yaml" + - "CONFIG_FILE=recipes/glm5.2/trtllm/gb300-fp4/agentx/variants.yaml:override_disagg_4p1d_dep8_c227_b16_mtp3" - "SLURM_PARTITION=batch_1" decode: num-worker: 1 @@ -7492,7 +7492,7 @@ qwen3.5-fp8-gb200-dynamo-sglang-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b128-c1x2x8-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_b128_c1x2x8_mtp" decode: num-worker: 1 tp: 4 @@ -7507,7 +7507,7 @@ qwen3.5-fp8-gb200-dynamo-sglang-mtp: ep: 8 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp8_ep8_d_tp8_ep8_b1024_c32x48x80_mtp" decode: num-worker: 1 tp: 8 @@ -7522,7 +7522,7 @@ qwen3.5-fp8-gb200-dynamo-sglang-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_3p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c480_mtp" decode: num-worker: 1 tp: 16 @@ -7537,7 +7537,7 @@ qwen3.5-fp8-gb200-dynamo-sglang-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_4p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c768_mtp" decode: num-worker: 1 tp: 16 @@ -7552,7 +7552,7 @@ qwen3.5-fp8-gb200-dynamo-sglang-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_6p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c1280_mtp" decode: num-worker: 1 tp: 16 @@ -7567,7 +7567,7 @@ qwen3.5-fp8-gb200-dynamo-sglang-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_7p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b2048_c1344_mtp" decode: num-worker: 1 tp: 16 @@ -7582,7 +7582,7 @@ qwen3.5-fp8-gb200-dynamo-sglang-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp8/8k1k/variants.yaml:override_disagg_8p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b2048_c1920x2304_mtp" decode: num-worker: 1 tp: 16 @@ -7619,7 +7619,7 @@ kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p1d-dcp8-dcp8-dspark4-mooncake.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p1d_dcp8_dcp8_dspark4_mooncake" decode: { num-worker: 1, tp: 8, dcp-size: 8, ep: 1, dp-attn: false } - spec-decoding: mtp kv-offloading: dram @@ -7632,7 +7632,7 @@ kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p2d-dcp8-dcp8-dspark4-mooncake.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p2d_dcp8_dcp8_dspark4_mooncake" decode: { num-worker: 2, tp: 8, dcp-size: 8, ep: 1, dp-attn: false } - spec-decoding: mtp kv-offloading: dram @@ -7645,7 +7645,7 @@ kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p3d-dcp8-dcp8-dspark7-mooncake.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p3d_dcp8_dcp8_dspark7_mooncake" decode: { num-worker: 3, tp: 8, dcp-size: 8, ep: 1, dp-attn: false } - spec-decoding: mtp kv-offloading: dram @@ -7658,7 +7658,7 @@ kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb300-fp4/agentx/disagg-1p3d-dcp8-dcp8-dspark4-mooncake.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml:override_disagg_1p3d_dcp8_dcp8_dspark4_mooncake" decode: { num-worker: 3, tp: 8, dcp-size: 8, ep: 1, dp-attn: false } kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-agg: @@ -7686,7 +7686,7 @@ kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb300-fp4/agentx/agg-dcp8-dspark7-maxseq2-mooncake.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml:override_agg_dcp8_dspark7_maxseq2_mooncake" - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: mooncake, version: "0.3.13.post1" } @@ -7699,7 +7699,7 @@ kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/gb300-fp4/agentx/agg-dcp8-dspark4-mooncake.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/gb300-fp4/agentx/variants.yaml:override_agg_dcp8_dspark4_mooncake" # Kimi-K3 MXFP4 B200 aggregated vLLM (TP8 x PP2, 2 nodes / 16 GPUs), agentic # coding. The native MXFP4 checkpoint (2.8T total params, ~1.4TB weights) does @@ -7740,7 +7740,7 @@ kimik3-fp4-b200-dynamo-vllm-agentic-dspark: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c1.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c1" - spec-decoding: mtp kv-offloading: none conc-list: [4] @@ -7753,7 +7753,7 @@ kimik3-fp4-b200-dynamo-vllm-agentic-dspark: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c4.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c4" - spec-decoding: mtp kv-offloading: none conc-list: [8] @@ -7766,7 +7766,7 @@ kimik3-fp4-b200-dynamo-vllm-agentic-dspark: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c8.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c8" - spec-decoding: mtp kv-offloading: none conc-list: [14] @@ -7779,7 +7779,7 @@ kimik3-fp4-b200-dynamo-vllm-agentic-dspark: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c14.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c14" - spec-decoding: mtp kv-offloading: none conc-list: [24] @@ -7792,7 +7792,7 @@ kimik3-fp4-b200-dynamo-vllm-agentic-dspark: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c24.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c24" - spec-decoding: mtp kv-offloading: none conc-list: [48] @@ -7805,7 +7805,7 @@ kimik3-fp4-b200-dynamo-vllm-agentic-dspark: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c48.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c48" - kv-offloading: none conc-list: [96] num-nodes: 2 @@ -7817,7 +7817,7 @@ kimik3-fp4-b200-dynamo-vllm-agentic-dspark: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c96.yaml" + - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/variants.yaml:override_agg_tp8pp2_mooncake_c96" qwen3.5-fp8-gb300-dynamo-sglang-mtp: image: lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 model: Qwen/Qwen3.5-397B-A17B-FP8 @@ -7843,7 +7843,7 @@ qwen3.5-fp8-gb300-dynamo-sglang-mtp: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_b1024_c1x2x8_mtp" decode: num-worker: 1 tp: 4 @@ -7858,7 +7858,7 @@ qwen3.5-fp8-gb300-dynamo-sglang-mtp: ep: 8 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_1p1d_p_tp8_ep8_d_tp8_ep8_b1024_c32x48x80_mtp" decode: num-worker: 1 tp: 8 @@ -7873,7 +7873,7 @@ qwen3.5-fp8-gb300-dynamo-sglang-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_3p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c480_mtp" decode: num-worker: 1 tp: 16 @@ -7888,7 +7888,7 @@ qwen3.5-fp8-gb300-dynamo-sglang-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_4p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c768_mtp" decode: num-worker: 1 tp: 16 @@ -7903,7 +7903,7 @@ qwen3.5-fp8-gb300-dynamo-sglang-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_6p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b1024_c1280_mtp" decode: num-worker: 1 tp: 16 @@ -7918,7 +7918,7 @@ qwen3.5-fp8-gb300-dynamo-sglang-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_7p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b2048_c1344_mtp" decode: num-worker: 1 tp: 16 @@ -7933,7 +7933,7 @@ qwen3.5-fp8-gb300-dynamo-sglang-mtp: ep: 4 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml" + - "CONFIG_FILE=recipes/qwen3.5/sglang/gb300-fp8/8k1k/variants.yaml:override_disagg_8p1d_p_tp4_ep4_dp4_d_tp16_ep16_dp16_b2048_c1920x2304_mtp" decode: num-worker: 1 tp: 16 @@ -8294,7 +8294,7 @@ dsv4-fp4-b200-dynamo-sglang-agentic-agg: additional-settings: - "SYNTHETIC_ACCEPTANCE=true" - "SYNTHETIC_ACCEPTANCE_LENGTH=3.77" - - "CONFIG_FILE=recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c1-mtp-hicache.yaml" + - "CONFIG_FILE=recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml:override_agg_b200_tp8_c1_mtp_hicache" - spec-decoding: mtp conc-list: [4] kv-offloading: dram @@ -8308,7 +8308,7 @@ dsv4-fp4-b200-dynamo-sglang-agentic-agg: additional-settings: - "SYNTHETIC_ACCEPTANCE=true" - "SYNTHETIC_ACCEPTANCE_LENGTH=3.77" - - "CONFIG_FILE=recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c4-mtp-hicache.yaml" + - "CONFIG_FILE=recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml:override_agg_b200_tp8_c4_mtp_hicache" - spec-decoding: mtp conc-list: [8] kv-offloading: dram @@ -8322,7 +8322,7 @@ dsv4-fp4-b200-dynamo-sglang-agentic-agg: additional-settings: - "SYNTHETIC_ACCEPTANCE=true" - "SYNTHETIC_ACCEPTANCE_LENGTH=3.77" - - "CONFIG_FILE=recipes/dsv4/sglang/b200-fp4/agentx/agg-b200-tp8-c8-mtp-hicache.yaml" + - "CONFIG_FILE=recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml:override_agg_b200_tp8_c8_mtp_hicache" dsv4-fp4-b200-dynamo-sglang-agentic-disagg: image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 @@ -8351,7 +8351,7 @@ dsv4-fp4-b200-dynamo-sglang-agentic-disagg: additional-settings: - "SYNTHETIC_ACCEPTANCE=true" - "SYNTHETIC_ACCEPTANCE_LENGTH=3.77" - - "CONFIG_FILE=recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml" + - "CONFIG_FILE=recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml:override_disagg_b200_1p1d_dep8_dep8_c64_mtp_kvoffload" decode: num-worker: 1 tp: 8 @@ -8369,7 +8369,7 @@ dsv4-fp4-b200-dynamo-sglang-agentic-disagg: additional-settings: - "SYNTHETIC_ACCEPTANCE=true" - "SYNTHETIC_ACCEPTANCE_LENGTH=3.77" - - "CONFIG_FILE=recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-1p1d-dep8-dep8-c128-mtp-kvoffload.yaml" + - "CONFIG_FILE=recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml:override_disagg_b200_1p1d_dep8_dep8_c128_mtp_kvoffload" decode: num-worker: 1 tp: 8 @@ -8387,7 +8387,7 @@ dsv4-fp4-b200-dynamo-sglang-agentic-disagg: additional-settings: - "SYNTHETIC_ACCEPTANCE=true" - "SYNTHETIC_ACCEPTANCE_LENGTH=3.77" - - "CONFIG_FILE=recipes/dsv4/sglang/b200-fp4/agentx/disagg-b200-2p1d-dep8-dep8-c256-mtp-kvoffload.yaml" + - "CONFIG_FILE=recipes/dsv4/sglang/b200-fp4/agentx/variants.yaml:override_disagg_b200_2p1d_dep8_dep8_c256_mtp_kvoffload" decode: num-worker: 1 tp: 8 @@ -8421,7 +8421,7 @@ qwen3.5-fp8-b200-dynamo-sglang-agentic-disagg-mtp: ep: 1 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c8-mtp.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_hicache_c8_mtp decode: num-worker: 1 tp: 4 @@ -8439,7 +8439,7 @@ qwen3.5-fp8-b200-dynamo-sglang-agentic-disagg-mtp: ep: 1 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c16-mtp.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_hicache_c16_mtp decode: num-worker: 1 tp: 4 @@ -8457,7 +8457,7 @@ qwen3.5-fp8-b200-dynamo-sglang-agentic-disagg-mtp: ep: 1 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c24-mtp.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_hicache_c24_mtp decode: num-worker: 1 tp: 4 @@ -8475,7 +8475,7 @@ qwen3.5-fp8-b200-dynamo-sglang-agentic-disagg-mtp: ep: 1 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c32-mtp.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_hicache_c32_mtp decode: num-worker: 1 tp: 4 @@ -8493,7 +8493,7 @@ qwen3.5-fp8-b200-dynamo-sglang-agentic-disagg-mtp: ep: 1 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c48-write-through-mtp.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_hicache_c48_write_through_mtp decode: num-worker: 1 tp: 4 @@ -8511,7 +8511,7 @@ qwen3.5-fp8-b200-dynamo-sglang-agentic-disagg-mtp: ep: 1 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b200-fp8/agentx/disagg-1p1d-p-tp4-d-tp4-hicache-c64-write-through-mtp.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b200-fp8/agentx/variants.yaml:override_disagg_1p1d_p_tp4_d_tp4_hicache_c64_write_through_mtp decode: num-worker: 1 tp: 4 @@ -8551,7 +8551,7 @@ qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg: ep: 1 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c4-mtp-hicache.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_colocated_c4_mtp_hicache decode: num-worker: 1 tp: 4 @@ -8569,7 +8569,7 @@ qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg: ep: 1 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c12-mtp-hicache.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_colocated_c12_mtp_hicache decode: num-worker: 1 tp: 4 @@ -8587,7 +8587,7 @@ qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg: ep: 1 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4-tp4-colocated-c24-mtp-hicache.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp4_tp4_colocated_c24_mtp_hicache decode: num-worker: 1 tp: 4 @@ -8605,7 +8605,7 @@ qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg: ep: 4 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp4ep4-tp4-colocated-c32-mtp-hicache.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp4ep4_tp4_colocated_c32_mtp_hicache decode: num-worker: 1 tp: 4 @@ -8623,7 +8623,7 @@ qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg: ep: 2 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c32-mtp-hicache.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c32_mtp_hicache decode: num-worker: 1 tp: 2 @@ -8641,7 +8641,7 @@ qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg: ep: 2 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c40-mtp-hicache.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c40_mtp_hicache decode: num-worker: 1 tp: 2 @@ -8659,7 +8659,7 @@ qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg: ep: 2 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c44-mtp-hicache.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c44_mtp_hicache decode: num-worker: 1 tp: 2 @@ -8677,7 +8677,7 @@ qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg: ep: 2 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c48-mtp-hicache.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c48_mtp_hicache decode: num-worker: 1 tp: 2 @@ -8695,7 +8695,7 @@ qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg: ep: 2 dp-attn: false additional-settings: - - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/disagg-1p1d-tp2ep2-tp2ep2-colocated-c56-replayssm-mtp-hicache.yaml + - CONFIG_FILE=recipes/qwen3.5/sglang/b300-fp8/agentx/variants.yaml:override_disagg_1p1d_tp2ep2_tp2ep2_colocated_c56_replayssm_mtp_hicache decode: num-worker: 1 tp: 2