From 6060acb61642d066c6727765767a0e2ec38043a8 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 14 Sep 2026 22:46:36 -0500 Subject: [PATCH 1/6] fix(amd): use Pro-0813 golden acceptance for AgentX (cherry picked from commit ebace4feb9ac06f641cdd08c86b95f3aeb7bcb03) --- .../multi_node/amd_utils/server_sglang.sh | 47 ++++++++++++++----- 1 file changed, 35 insertions(+), 12 deletions(-) diff --git a/benchmarks/multi_node/amd_utils/server_sglang.sh b/benchmarks/multi_node/amd_utils/server_sglang.sh index b44ea1a11f..286f675c81 100755 --- a/benchmarks/multi_node/amd_utils/server_sglang.sh +++ b/benchmarks/multi_node/amd_utils/server_sglang.sh @@ -1378,18 +1378,41 @@ else echo "[INFO] Eval mode: synthetic MTP disabled (using real acceptance)" else DSV4_GOLDEN_AL="" - case "${MODEL_NAME}:${DECODE_MTP_SIZE}" in - DeepSeek-V4-Pro-0813:1) DSV4_GOLDEN_AL=1.84 ;; - DeepSeek-V4-Pro-0813:2) DSV4_GOLDEN_AL=2.51 ;; - DeepSeek-V4-Pro-0813:3) DSV4_GOLDEN_AL=3.01 ;; - DeepSeek-V4-Pro-0813:*) - echo "ERROR: Pro-0813 draft length ${DECODE_MTP_SIZE} has no golden AL wired here; refusing to use the original V4 curve." >&2 - exit 1 - ;; - *DeepSeek-V4*:1) DSV4_GOLDEN_AL=1.79 ;; - *DeepSeek-V4*:2) DSV4_GOLDEN_AL=2.27 ;; - *DeepSeek-V4*:3) DSV4_GOLDEN_AL=2.49 ;; - esac + if [[ "${SPEC_DECODING:-}" == "draft_model" ]]; then + # DSpark curve, keyed by block size gamma. Source: + # golden_al_distribution/dsv4-pro-0813-dspark.yaml (thinking_on). + # It peaks at gamma 6 and regresses past it, so 7 and 8 are listed + # to keep an accidental over-long block from silently falling back + # to real acceptance. + case "${MODEL_NAME}:${DECODE_MTP_SIZE}" in + *DeepSeek-V4-Pro-0813*:1) DSV4_GOLDEN_AL=1.84 ;; + *DeepSeek-V4-Pro-0813*:2) DSV4_GOLDEN_AL=2.51 ;; + *DeepSeek-V4-Pro-0813*:3) DSV4_GOLDEN_AL=3.01 ;; + *DeepSeek-V4-Pro-0813*:4) DSV4_GOLDEN_AL=3.36 ;; + *DeepSeek-V4-Pro-0813*:5) DSV4_GOLDEN_AL=3.61 ;; + *DeepSeek-V4-Pro-0813*:6) DSV4_GOLDEN_AL=3.77 ;; + *DeepSeek-V4-Pro-0813*:7) DSV4_GOLDEN_AL=3.73 ;; + *DeepSeek-V4-Pro-0813*:8) DSV4_GOLDEN_AL=3.47 ;; + esac + else + # EAGLE/MTP path. Pro-0813 draws on the same committed + # thinking-on curve, so only the lengths calibrated there are + # wired; an unwired length is an error rather than a silent + # fall-through to the original V4 curve below, which would + # understate acceptance (2.49 against 3.01 at length 3). + case "${MODEL_NAME}:${DECODE_MTP_SIZE}" in + *DeepSeek-V4-Pro-0813*:1) DSV4_GOLDEN_AL=1.84 ;; + *DeepSeek-V4-Pro-0813*:2) DSV4_GOLDEN_AL=2.51 ;; + *DeepSeek-V4-Pro-0813*:3) DSV4_GOLDEN_AL=3.01 ;; + *DeepSeek-V4-Pro-0813*:*) + echo "ERROR: Pro-0813 draft length ${DECODE_MTP_SIZE} has no golden AL wired here; refusing to use the original V4 curve." >&2 + exit 1 + ;; + *DeepSeek-V4*:1) DSV4_GOLDEN_AL=1.79 ;; + *DeepSeek-V4*:2) DSV4_GOLDEN_AL=2.27 ;; + *DeepSeek-V4*:3) DSV4_GOLDEN_AL=2.49 ;; + esac + fi if [[ -n "$DSV4_GOLDEN_AL" ]]; then DECODE_SIM_ACC_ENV="SGLANG_SIMULATE_ACC_LEN=${DSV4_GOLDEN_AL} SGLANG_SIMULATE_ACC_METHOD=match-expected SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token" else From 8fa6ee56f03a3694bb9dafabac76e48295ecc686 Mon Sep 17 00:00:00 2001 From: Theresa Shan Date: Tue, 15 Sep 2026 02:01:57 +0000 Subject: [PATCH 2/6] enable dspark Signed-off-by: Theresa Shan --- benchmarks/multi_node/amd_utils/models.yaml | 12 +++++++++-- .../multi_node/amd_utils/server_sglang.sh | 10 +++++++++- configs/amd-master.yaml | 20 +++++++++++-------- 3 files changed, 31 insertions(+), 11 deletions(-) diff --git a/benchmarks/multi_node/amd_utils/models.yaml b/benchmarks/multi_node/amd_utils/models.yaml index f8507cf0f0..85eb0ca049 100644 --- a/benchmarks/multi_node/amd_utils/models.yaml +++ b/benchmarks/multi_node/amd_utils/models.yaml @@ -8,7 +8,8 @@ # Schema: # : # base_flags: str # Common flags for both prefill and decode -# mtp_flags: str # Appended to decode when DECODE_MTP_SIZE > 0 +# mtp_flags: str # Appended when DECODE_MTP_SIZE > 0 and spec-decoding is mtp +# dspark_flags: str # Replaces mtp_flags when spec-decoding is draft_model # dp_flags: str # Appended when DP attention is enabled (prefill or decode) # ep_flags: str # Appended when EP is enabled. EP-specific MoE knobs only # # (a2a backend, deepep mode, ep-dispatch algorithm). With @@ -354,9 +355,16 @@ DeepSeek-R1-0528-MXFP4-v2: DeepSeek-V4-Pro-AgentX: &DeepSeek-V4-Pro-AgentX base_flags: "--enable-deepseek-v4-fp4-indexer --watchdog-timeout 3600 --load-balance-method round_robin --kv-cache-dtype fp8_e4m3 --attention-backend dsv4 --page-size 256 --swa-full-tokens-ratio 0.1 --enforce-shared-experts-fusion --tool-call-parser deepseekv4 --reasoning-parser deepseek-v4 --disaggregation-transfer-backend mori --tokenizer-worker-num 8 --stream-interval 20 --log-level info --log-level-http error" - dp_flags: "--enable-dp-attention --swa-full-tokens-ratio 0.15 --enable-dp-attention-local-control-broadcast" + dp_flags: "--enable-dp-attention --enable-dp-lm-head --swa-full-tokens-ratio 0.15 --enable-dp-attention-local-control-broadcast" ep_flags: "--ep-dispatch-algorithm fake --moe-a2a-backend mori --deepep-mode normal" mtp_flags: "--speculative-algorithm EAGLE --speculative-eagle-topk 1" + # Selected instead of mtp_flags when the arm sets spec-decoding: draft_model. + # The DSpark draft head ships inside the DeepSeek-V4-Pro-0813 target checkpoint + # (dspark_block_size / dspark_markov_rank / dspark_target_layer_ids in + # config.json), so --speculative-draft-model-path defaults to --model-path and + # no separate draft checkpoint is needed. DECODE_MTP_SIZE carries the block + # size gamma; server_sglang.sh derives the verify window as gamma + 1. + dspark_flags: "--speculative-algorithm DSPARK --speculative-eagle-topk 1" prefill: disable_radix_cache: false disable_cuda_graph: true diff --git a/benchmarks/multi_node/amd_utils/server_sglang.sh b/benchmarks/multi_node/amd_utils/server_sglang.sh index 286f675c81..61b2302061 100755 --- a/benchmarks/multi_node/amd_utils/server_sglang.sh +++ b/benchmarks/multi_node/amd_utils/server_sglang.sh @@ -100,6 +100,7 @@ def parse_range(cuda_range, default_start, default_end): # Output shell variables print(f'MODEL_BASE_FLAGS=\"{m.get(\"base_flags\", \"\")}\"') print(f'MODEL_MTP_FLAGS=\"{m.get(\"mtp_flags\", \"\")}\"') +print(f'MODEL_DSPARK_FLAGS=\"{m.get(\"dspark_flags\", \"\")}\"') print(f'MODEL_DP_FLAGS=\"{m.get(\"dp_flags\", \"\")}\"') print(f'MODEL_EP_FLAGS=\"{m.get(\"ep_flags\", \"\")}\"') @@ -382,7 +383,14 @@ build_server_config() { local specific_config="" if [ "$decode_mtp_size" -gt 0 ]; then - mtp_config="${MODEL_MTP_FLAGS} --speculative-num-steps ${decode_mtp_size} --speculative-num-draft-tokens $((decode_mtp_size + 1))" + if [[ "${SPEC_DECODING:-}" == "draft_model" ]]; then + # DSpark proposes a whole block per step rather than walking a chain, + # so num-steps is pinned to 1 and decode_mtp_size is read as the block + # size gamma. The verify window is gamma + 1, same arithmetic as MTP. + mtp_config="${MODEL_DSPARK_FLAGS} --speculative-dspark-block-size ${decode_mtp_size} --speculative-num-steps 1 --speculative-num-draft-tokens $((decode_mtp_size + 1))" + else + mtp_config="${MODEL_MTP_FLAGS} --speculative-num-steps ${decode_mtp_size} --speculative-num-draft-tokens $((decode_mtp_size + 1))" + fi fi if [[ "$enable_dp" == "true" ]]; then diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index e1d48355a8..2c94f947b6 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -1207,7 +1207,7 @@ dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: agentic-coding: - dram-utilization: 0.80 search-space: - - spec-decoding: "mtp" + - spec-decoding: "draft_model" conc-list: [ 4 ] kv-offloading: none prefill: @@ -1225,8 +1225,8 @@ dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: dp-attn: false additional-settings: - "DECODE_NODES=1" - - "DECODE_MTP_SIZE=3" - - spec-decoding: "mtp" + - "DECODE_MTP_SIZE=6" + - spec-decoding: "draft_model" conc-list: [ 16 ] kv-offloading: none prefill: @@ -1244,8 +1244,8 @@ dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: dp-attn: false additional-settings: - "DECODE_NODES=1" - - "DECODE_MTP_SIZE=3" - - spec-decoding: "mtp" + - "DECODE_MTP_SIZE=6" + - spec-decoding: "draft_model" conc-list: [ 32, 48 ] kv-offloading: dram kv-offload-backend: { name: hicache } @@ -1265,8 +1265,12 @@ dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: dp-attn: false additional-settings: - "DECODE_NODES=1" - - "DECODE_MTP_SIZE=3" - - spec-decoding: "mtp" + # DSpark block size gamma, not an MTP chain length: the verify window + # is gamma + 1. 6 is the peak of the committed golden AL curve + # (golden_al_distribution/dsv4-pro-0813-dspark.yaml: 6 -> 3.77), which + # regresses at 7 and 8. + - "DECODE_MTP_SIZE=6" + - spec-decoding: "draft_model" conc-list: [ 128, 192, 256 ] kv-offloading: dram kv-offload-backend: { name: umbp-linker } @@ -1292,7 +1296,7 @@ dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: dp-attn: true additional-settings: - "DECODE_NODES=1" - - "DECODE_MTP_SIZE=3" + - "DECODE_MTP_SIZE=6" minimaxm3-fp4-mi355x-vllm-agentic-mtp: image: vllm/vllm-openai-rocm:nightly-2a02f6efe319c885e3ccbcecde402e0028f9ec1e From a6a97c3e0fd073a30f74bdacdc3c2f2f714afff0 Mon Sep 17 00:00:00 2001 From: Theresa Shan Date: Wed, 16 Sep 2026 14:36:53 +0000 Subject: [PATCH 3/6] align config and image with intenal sweep Signed-off-by: Theresa Shan --- .../multi_node/amd_utils/server_sglang.sh | 55 +++++++------------ configs/amd-master.yaml | 14 ++--- 2 files changed, 25 insertions(+), 44 deletions(-) diff --git a/benchmarks/multi_node/amd_utils/server_sglang.sh b/benchmarks/multi_node/amd_utils/server_sglang.sh index 61b2302061..aae2a7f917 100755 --- a/benchmarks/multi_node/amd_utils/server_sglang.sh +++ b/benchmarks/multi_node/amd_utils/server_sglang.sh @@ -384,6 +384,10 @@ build_server_config() { if [ "$decode_mtp_size" -gt 0 ]; then if [[ "${SPEC_DECODING:-}" == "draft_model" ]]; then + if [[ -z "${MODEL_DSPARK_FLAGS// }" ]]; then + echo "FATAL: SPEC_DECODING=draft_model but model '${model_name}' has no dspark_flags in models.yaml." >&2 + exit 1 + fi # DSpark proposes a whole block per step rather than walking a chain, # so num-steps is pinned to 1 and decode_mtp_size is read as the block # size gamma. The verify window is gamma + 1, same arithmetic as MTP. @@ -1386,41 +1390,22 @@ else echo "[INFO] Eval mode: synthetic MTP disabled (using real acceptance)" else DSV4_GOLDEN_AL="" - if [[ "${SPEC_DECODING:-}" == "draft_model" ]]; then - # DSpark curve, keyed by block size gamma. Source: - # golden_al_distribution/dsv4-pro-0813-dspark.yaml (thinking_on). - # It peaks at gamma 6 and regresses past it, so 7 and 8 are listed - # to keep an accidental over-long block from silently falling back - # to real acceptance. - case "${MODEL_NAME}:${DECODE_MTP_SIZE}" in - *DeepSeek-V4-Pro-0813*:1) DSV4_GOLDEN_AL=1.84 ;; - *DeepSeek-V4-Pro-0813*:2) DSV4_GOLDEN_AL=2.51 ;; - *DeepSeek-V4-Pro-0813*:3) DSV4_GOLDEN_AL=3.01 ;; - *DeepSeek-V4-Pro-0813*:4) DSV4_GOLDEN_AL=3.36 ;; - *DeepSeek-V4-Pro-0813*:5) DSV4_GOLDEN_AL=3.61 ;; - *DeepSeek-V4-Pro-0813*:6) DSV4_GOLDEN_AL=3.77 ;; - *DeepSeek-V4-Pro-0813*:7) DSV4_GOLDEN_AL=3.73 ;; - *DeepSeek-V4-Pro-0813*:8) DSV4_GOLDEN_AL=3.47 ;; - esac - else - # EAGLE/MTP path. Pro-0813 draws on the same committed - # thinking-on curve, so only the lengths calibrated there are - # wired; an unwired length is an error rather than a silent - # fall-through to the original V4 curve below, which would - # understate acceptance (2.49 against 3.01 at length 3). - case "${MODEL_NAME}:${DECODE_MTP_SIZE}" in - *DeepSeek-V4-Pro-0813*:1) DSV4_GOLDEN_AL=1.84 ;; - *DeepSeek-V4-Pro-0813*:2) DSV4_GOLDEN_AL=2.51 ;; - *DeepSeek-V4-Pro-0813*:3) DSV4_GOLDEN_AL=3.01 ;; - *DeepSeek-V4-Pro-0813*:*) - echo "ERROR: Pro-0813 draft length ${DECODE_MTP_SIZE} has no golden AL wired here; refusing to use the original V4 curve." >&2 - exit 1 - ;; - *DeepSeek-V4*:1) DSV4_GOLDEN_AL=1.79 ;; - *DeepSeek-V4*:2) DSV4_GOLDEN_AL=2.27 ;; - *DeepSeek-V4*:3) DSV4_GOLDEN_AL=2.49 ;; - esac - fi + case "${MODEL_NAME}:${DECODE_MTP_SIZE}" in + DeepSeek-V4-Pro-0813:1) DSV4_GOLDEN_AL=1.84 ;; + DeepSeek-V4-Pro-0813:2) DSV4_GOLDEN_AL=2.51 ;; + DeepSeek-V4-Pro-0813:3) DSV4_GOLDEN_AL=3.01 ;; + # The dspark curve does carry values out to length 8, but only + # 1-3 are wired here, matching upstream. Refuse rather than fall + # through to the original V4 curve, which would silently + # under-simulate this checkpoint. + DeepSeek-V4-Pro-0813:*) + echo "ERROR: Pro-0813 draft length ${DECODE_MTP_SIZE} has no golden AL wired here; refusing to use the original V4 curve." >&2 + exit 1 + ;; + *DeepSeek-V4*:1) DSV4_GOLDEN_AL=1.79 ;; + *DeepSeek-V4*:2) DSV4_GOLDEN_AL=2.27 ;; + *DeepSeek-V4*:3) DSV4_GOLDEN_AL=2.49 ;; + esac if [[ -n "$DSV4_GOLDEN_AL" ]]; then DECODE_SIM_ACC_ENV="SGLANG_SIMULATE_ACC_LEN=${DSV4_GOLDEN_AL} SGLANG_SIMULATE_ACC_METHOD=match-expected SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token" else diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index 2c94f947b6..fdcd390fbe 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -1194,7 +1194,7 @@ minimaxm3-fp8-mi325x-vllm-agentic-mtp: - { tp: 8, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 2, 4, 8, 10, 12, 14, 16, 18] } dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: - image: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260911 + image: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260913 model: deepseek-ai/DeepSeek-V4-Pro-0813 model-prefix: dsv4 runner: cluster:mi355x-amds @@ -1225,7 +1225,7 @@ dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: dp-attn: false additional-settings: - "DECODE_NODES=1" - - "DECODE_MTP_SIZE=6" + - "DECODE_MTP_SIZE=3" - spec-decoding: "draft_model" conc-list: [ 16 ] kv-offloading: none @@ -1244,7 +1244,7 @@ dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: dp-attn: false additional-settings: - "DECODE_NODES=1" - - "DECODE_MTP_SIZE=6" + - "DECODE_MTP_SIZE=3" - spec-decoding: "draft_model" conc-list: [ 32, 48 ] kv-offloading: dram @@ -1265,11 +1265,7 @@ dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: dp-attn: false additional-settings: - "DECODE_NODES=1" - # DSpark block size gamma, not an MTP chain length: the verify window - # is gamma + 1. 6 is the peak of the committed golden AL curve - # (golden_al_distribution/dsv4-pro-0813-dspark.yaml: 6 -> 3.77), which - # regresses at 7 and 8. - - "DECODE_MTP_SIZE=6" + - "DECODE_MTP_SIZE=3" - spec-decoding: "draft_model" conc-list: [ 128, 192, 256 ] kv-offloading: dram @@ -1296,7 +1292,7 @@ dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: dp-attn: true additional-settings: - "DECODE_NODES=1" - - "DECODE_MTP_SIZE=6" + - "DECODE_MTP_SIZE=3" minimaxm3-fp4-mi355x-vllm-agentic-mtp: image: vllm/vllm-openai-rocm:nightly-2a02f6efe319c885e3ccbcecde402e0028f9ec1e From ce07190aac1cdff468bb0483390f7d0efd4d9440 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Wed, 16 Sep 2026 03:02:26 -0500 Subject: [PATCH 4/6] fix(amd): forward required AgentX client policy inputs --- benchmarks/runtime_settings.sh | 2 +- perf-changelog.yaml | 10 ++++++++++ 2 files changed, 11 insertions(+), 1 deletion(-) diff --git a/benchmarks/runtime_settings.sh b/benchmarks/runtime_settings.sh index 0b04efe9ba..a369898f8b 100644 --- a/benchmarks/runtime_settings.sh +++ b/benchmarks/runtime_settings.sh @@ -32,4 +32,4 @@ export SGLANG_TORCH_PROFILER_DIR='/workspace' export VLLM_TORCH_PROFILER_DIR='/workspace' # Explicitly forward these settings across container boundaries. -export INFERENCEX_RUNTIME_ENV_VARS="OPENAI_API_KEY SWEBENCH_EXPECTED_INSTANCES SWEBENCH_AGENT_STEP_LIMIT SWEBENCH_AGENT_TIMEOUT SWEBENCH_AGENT_EXIT_GRACE SWEBENCH_WATCHDOG_POLL SWEBENCH_SANDBOX_SWEEP SWEBENCH_SKIP_SCORE SWEBENCH_EVAL_TIMEOUT SWEBENCH_SCORE_TIMEOUT SWEBENCH_MAX_WORKERS EVAL_ENDPOINT_READY_TIMEOUT_SECONDS EVAL_MODEL_STABILIZATION_SECONDS AIPERF_FAILED_REQUEST_THRESHOLD AIPERF_LIVE_FAILED_REQUEST_THRESHOLD AIPERF_TRACE_IDLE_GAP_CAP_SECONDS AIPERF_PYTHON_VERSION AIPERF_WARMUP_REQUESTS_PER_LANE AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES AGENTIC_WARMUP_GRACE_PERIOD AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS AIPERF_UNSAFE_OVERRIDE ENABLE_AGENTX_POWER VLLM_ENGINE_READY_TIMEOUT_S SGLANG_TORCH_PROFILER_DIR VLLM_TORCH_PROFILER_DIR" +export INFERENCEX_RUNTIME_ENV_VARS="OPENAI_API_KEY SWEBENCH_EXPECTED_INSTANCES SWEBENCH_AGENT_STEP_LIMIT SWEBENCH_AGENT_TIMEOUT SWEBENCH_AGENT_EXIT_GRACE SWEBENCH_WATCHDOG_POLL SWEBENCH_SANDBOX_SWEEP SWEBENCH_SKIP_SCORE SWEBENCH_EVAL_TIMEOUT SWEBENCH_SCORE_TIMEOUT SWEBENCH_MAX_WORKERS EVAL_ENDPOINT_READY_TIMEOUT_SECONDS EVAL_MODEL_STABILIZATION_SECONDS AIPERF_FAILED_REQUEST_THRESHOLD AIPERF_LIVE_FAILED_REQUEST_THRESHOLD AIPERF_TRACE_IDLE_GAP_CAP_SECONDS AIPERF_PYTHON_VERSION AIPERF_WARMUP_REQUESTS_PER_LANE AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES AGENTIC_WARMUP_GRACE_PERIOD AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS AIPERF_EXPERIMENTAL_FAST AIPERF_UNSAFE_OVERRIDE ENABLE_AGENTX_POWER REQUIRE_POWER VLLM_ENGINE_READY_TIMEOUT_S SGLANG_TORCH_PROFILER_DIR VLLM_TORCH_PROFILER_DIR" diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 8d11fa35b6..e560adf1e2 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -7923,3 +7923,13 @@ - "Bump image to lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260915." - "Switch HiCache defaults to --hicache-io-backend kernel and --hicache-mem-layout page_first (from direct / page_first_direct). Ratio 1.5 and write_through are unchanged." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3118 + +- config-keys: + - dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp + scenario-type: + - agentic-coding + description: + - "Rerun the MI355X disaggregated MTP sweep with the original DeepSeek-V4-Pro checkpoint, whose native MTP weights match the existing EAGLE recipe. The existing checkpoint selector restores thinking-on golden AL 2.49 for three draft tokens; evals use real acceptance." + - "Collect completed server logs from every Slurm node into the shared artifact directory after the benchmark server step exits, including decode-node logs." + - "Pass the existing router port to evaluation and forward workflow-owned AgentX fast-mode and power requirements across container boundaries." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3170 From 5a5ad970de5b53daa2cc5db0848e57091ee5c4c9 Mon Sep 17 00:00:00 2001 From: Theresa Shan Date: Wed, 16 Sep 2026 14:47:07 +0000 Subject: [PATCH 5/6] update perf-changelog.yaml Signed-off-by: Theresa Shan --- perf-changelog.yaml | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index e560adf1e2..339f55e017 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -7929,7 +7929,6 @@ scenario-type: - agentic-coding description: - - "Rerun the MI355X disaggregated MTP sweep with the original DeepSeek-V4-Pro checkpoint, whose native MTP weights match the existing EAGLE recipe. The existing checkpoint selector restores thinking-on golden AL 2.49 for three draft tokens; evals use real acceptance." - - "Collect completed server logs from every Slurm node into the shared artifact directory after the benchmark server step exits, including decode-node logs." - - "Pass the existing router port to evaluation and forward workflow-owned AgentX fast-mode and power requirements across container boundaries." + - "Bump image to lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260913." + - "Rerun the MI355X disaggregated MTP sweep with DeepSeek-V4-Pro-0813 + Dspark." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3170 From 329e7ea7cbcbf5baf248e5245d2c6dc9a3e0aaaf Mon Sep 17 00:00:00 2001 From: Theresa Shan Date: Wed, 16 Sep 2026 14:54:45 +0000 Subject: [PATCH 6/6] docs: point perf-changelog pr-link to #3191 Correct the pr-link picked up from the cherry-picked #3170 entry now that this DeepSeek-V4-Pro-0813 DSpark change has its own PR. Signed-off-by: Theresa Shan Co-authored-by: Cursor --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 339f55e017..d0e525974e 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -7931,4 +7931,4 @@ description: - "Bump image to lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260913." - "Rerun the MI355X disaggregated MTP sweep with DeepSeek-V4-Pro-0813 + Dspark." - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3170 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3191